From d804ea8a797cd63156064f156f1be1f54d0be6a4 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 21:43:20 +0000 Subject: [PATCH 01/11] deepnsm-v2 bible_wave: counted PoS pick replaces word_forms first-wins (D-LXC-1) --- .claude/settings.json | 4 + crates/deepnsm-v2/Cargo.toml | 7 + crates/deepnsm-v2/examples/bible_wave.rs | 297 +++++++++++++++++++++-- 3 files changed, 294 insertions(+), 14 deletions(-) diff --git a/.claude/settings.json b/.claude/settings.json index 7ba2f4401..98ea7be10 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -1,5 +1,9 @@ { "$schema": "https://json.schemastore.org/claude-code-settings.json", + "attribution": { + "commit": "", + "pr": "" + }, "permissions": { "allow": [ "Edit(**/*.md)", diff --git a/crates/deepnsm-v2/Cargo.toml b/crates/deepnsm-v2/Cargo.toml index b2fd37f67..aaf22ba67 100644 --- a/crates/deepnsm-v2/Cargo.toml +++ b/crates/deepnsm-v2/Cargo.toml @@ -27,3 +27,10 @@ crate is untouched; this is the parallel v2. # `identity_v2()` (bytes 14..16) cannot read them back. Every key this crate # minted before 2026-08-22 was a V1 key for that reason. lance-graph-contract = { path = "../lance-graph-contract", features = ["guid-v3-tail"] } + +# `test = true` makes `cargo test` run the example's own `#[cfg(test)]` tests +# (the D-LXC-1 tagger). Cargo only compiles an example by default; its `main()` +# never runs under test, so the KJV text is not needed. +[[example]] +name = "bible_wave" +test = true diff --git a/crates/deepnsm-v2/examples/bible_wave.rs b/crates/deepnsm-v2/examples/bible_wave.rs index 572e09483..54d369fb2 100644 --- a/crates/deepnsm-v2/examples/bible_wave.rs +++ b/crates/deepnsm-v2/examples/bible_wave.rs @@ -12,8 +12,9 @@ //! cargo run --example bible_wave -- /path/to/pg10.txt //! ``` //! -//! Pipeline: verses → PoS-tag (COCA lemma lexicon + documented archaic -//! fallback) → FSM → SPO stream (verse index = version) → `TemporalStream` + +//! Pipeline: verses → PoS-tag (COCA lemma table, then the counted +//! `word_forms.csv` evidence, then a documented archaic fallback) → FSM → SPO +//! stream (verse index = version) → `TemporalStream` + //! the TRAINED Cam96 codebook (`data/`, real Jina-v3 embeddings). //! //! Gates (panic on KILL): @@ -27,8 +28,9 @@ //! horizon → the Escalate zone). use deepnsm_v2::{ - load_cam96_codes, load_cam96_space, parse_to_spo, Nsm, PaletteVocab, Pos, Spo, Tagged, - TemporalStream, + load_cam96_codes, load_cam96_space, load_word_forms_csv, parse_to_spo, EvidenceError, + LexicalEvidence, LexicalReading, Nsm, PaletteVocab, Pos, Spo, Tagged, TemporalStream, WordId, + WordFormsReport, }; use std::collections::HashMap; use std::path::PathBuf; @@ -135,7 +137,34 @@ fn main() { "/../deepnsm/word_frequency/word_forms.csv" )) .expect("word_forms.csv (sibling deepnsm crate)"); - let pos_of = load_pos(&lemmas_csv, &forms_csv); + let tagger = Tagger::load(&lemmas_csv, &forms_csv, &nsm.vocab).expect("word_forms.csv evidence"); + let r = &tagger.report; + println!( + "LEXICON word_forms: {} rows, {} readings stored, {} empty surface, {} not in vocab", + r.rows, r.stored, r.empty_surface, r.unrouted + ); + // G6 (D-LXC-1) — how many in-vocabulary tags the counted pick moves away + // from the old first-wins rule. Pinned against the released + // `bible_vocab.txt`: any other number means the tables or the rule changed. + let first_wins = load_pos_first_wins(&lemmas_csv, &forms_csv); + let mut moved = 0usize; + for id in 0..nsm.vocab.len() { + let id = WordId::try_from(id).expect("vocab fits u16"); + let Some(w) = nsm.vocab.word(id) else { continue }; + let old = first_wins + .get(w) + .copied() + .or_else(|| archaic_pos(w)) + .unwrap_or(Pos::Other); + if tagger.pos(w, id).expect("count sum") != old { + moved += 1; + } + } + assert_eq!( + moved, 25, + "KILL G6: the counted pick moved {moved} in-vocabulary tags, pinned 25" + ); + println!("G6 PASS counted pick moves {moved} in-vocabulary tags from first-wins (pinned 25)"); // ── stream: verse index = version; FSM → SPO ── let mut stream = TemporalStream::new(); @@ -146,11 +175,7 @@ fn main() { for tok in verse.split_whitespace() { let Some(w) = normalise(tok) else { continue }; let Some(id) = nsm.vocab.id(&w) else { continue }; - let pos = pos_of - .get(&w) - .copied() - .or_else(|| archaic_pos(&w)) - .unwrap_or(Pos::Other); + let pos = tagger.pos(&w, id).expect("count sum"); tagged_buf.push(Tagged::new(id, pos)); } tagged_buf.push(Tagged::new(0, Pos::Stop)); // verse boundary flushes @@ -1017,10 +1042,116 @@ fn normalise(tok: &str) -> Option { (w.len() >= 2).then_some(w) } -/// `word -> Pos` from the COCA lemma table, with `word_forms.csv` layered UNDER -/// it so inflected forms (`created`, `made`) are not lost to `Pos::Other`. -/// Lemma-first, so no word that already had a tag gets a different one. -fn load_pos(lemmas_csv: &str, forms_csv: &str) -> HashMap { +/// The fold's states in tie-break order: a tie between two known sums goes to +/// the state listed first. +const PICK_ORDER: [Pos; 5] = [Pos::Noun, Pos::Verb, Pos::Adj, Pos::Det, Pos::Other]; + +/// The counted reading of one word (D-LXC-1). +/// +/// Folds each reading's source tag with [`coca_pos`] and sums the known +/// `wordFreq` per parser state. A state with any unknown count is unknown and +/// does not compete; overflow is an error, never a wrap. The highest known sum +/// wins, ties go to [`PICK_ORDER`], and no known count at all is `None` — the +/// caller's fallback decides, nothing is invented here. +/// +/// This replaces taking the FIRST `word_forms.csv` row: that file is ordered by +/// lemma rank, not by surface frequency, so for 259 surfaces the first row is +/// not the dominant reading (`changes`: verb row 13,624 first, noun 113,085). +fn counted_pos(readings: &[LexicalReading]) -> Result, EvidenceError> { + let mut best: Option<(u64, Pos)> = None; + for state in PICK_ORDER { + let mut sum = Some(0u64); + let mut seen = false; + for r in readings { + if coca_pos(&r.pos.as_char().to_string()) != state { + continue; + } + seen = true; + sum = match (sum, r.form_count) { + (Some(s), Some(c)) => Some(s.checked_add(c).ok_or(EvidenceError::CountOverflow)?), + _ => None, + }; + } + let (true, Some(sum)) = (seen, sum) else { + continue; + }; + // Strictly greater: an equal sum keeps the earlier state in PICK_ORDER. + if best.is_none_or(|(b, _)| sum > b) { + best = Some((sum, state)); + } + } + Ok(best.map(|(_, p)| p)) +} + +/// Lowercase the `word` column of `word_forms.csv`, leaving the header and the +/// other fields as they are. `load_word_forms_csv` matches surfaces exactly and +/// the corpus tokens are lowercased; three COCA surfaces (`True`, `False`, +/// `reElection`) would otherwise never route. +fn lowercase_word_column(forms_csv: &str) -> String { + let mut out = String::with_capacity(forms_csv.len()); + for (i, line) in forms_csv.lines().enumerate() { + if i > 0 { + if let Some((head, word)) = line.rsplit_once(',') { + out.push_str(head); + out.push(','); + out.push_str(&word.to_lowercase()); + out.push('\n'); + continue; + } + } + out.push_str(line); + out.push('\n'); + } + out +} + +/// The corpus tagger: lemma table → counted `word_forms.csv` evidence → +/// [`archaic_pos`] → [`Pos::Other`]. +/// +/// Lemma-first is deliberate and pinned (commit `ec50f07b`): the forms layer +/// only fills silence, never overrules a word the lemma table knew. +struct Tagger { + lemmas: HashMap, + evidence: LexicalEvidence, + report: WordFormsReport, +} + +impl Tagger { + fn load(lemmas_csv: &str, forms_csv: &str, vocab: &PaletteVocab) -> Result { + let mut lemmas = HashMap::new(); + for line in lemmas_csv.lines().skip(1) { + let f: Vec<&str> = line.split(',').collect(); + let (Some(lemma), Some(pos)) = (f.get(1), f.get(2)) else { + continue; + }; + lemmas + .entry(lemma.to_lowercase()) + .or_insert_with(|| coca_pos(pos)); + } + let (evidence, report) = load_word_forms_csv(&lowercase_word_column(forms_csv), vocab)?; + Ok(Self { + lemmas, + evidence, + report, + }) + } + + /// The tag for word `w` whose routing id is `id`. + fn pos(&self, w: &str, id: WordId) -> Result { + if let Some(&p) = self.lemmas.get(w) { + return Ok(p); + } + if let Some(p) = counted_pos(self.evidence.readings(id))? { + return Ok(p); + } + Ok(archaic_pos(w).unwrap_or(Pos::Other)) + } +} + +/// The tagging `bible_wave` used before D-LXC-1: the lemma table, then the +/// FIRST `word_forms.csv` row per surface. Kept only so gate G6 can count how +/// many tags the counted pick moves. +fn load_pos_first_wins(lemmas_csv: &str, forms_csv: &str) -> HashMap { let mut m: HashMap = HashMap::new(); for line in lemmas_csv.lines().skip(1) { let f: Vec<&str> = line.split(',').collect(); @@ -1040,3 +1171,141 @@ fn load_pos(lemmas_csv: &str, forms_csv: &str) -> HashMap { } m } + +// D-LXC-1 tests. They run under `cargo test` because `Cargo.toml` declares this +// example with `test = true`; cargo never runs an example's `main()`, so the +// KJV itself is not needed here. +#[cfg(test)] +mod tests { + use super::*; + use deepnsm_v2::PosCode; + + fn reading(tag: u8, count: Option) -> LexicalReading { + LexicalReading { + pos: PosCode(tag), + lemma: None, + form_count: count, + } + } + + fn vocab(words: &[&str]) -> PaletteVocab { + let mut v = PaletteVocab::new(); + v.from_frequency_ranked(words.iter().copied()); + v + } + + fn committed(name: &str) -> String { + std::fs::read_to_string(format!( + "{}/../deepnsm/word_frequency/{name}", + env!("CARGO_MANIFEST_DIR") + )) + .expect("committed COCA table") + } + + // (a) a homograph keeps every reading + #[test] + fn record_keeps_its_noun_and_verb_readings() { + let v = vocab(&["record"]); + let t = Tagger::load("rank,lemma,PoS\n", &committed("word_forms.csv"), &v).unwrap(); + let tags: Vec = t + .evidence + .readings(v.id("record").unwrap()) + .iter() + .map(|r| r.pos.as_char()) + .collect(); + assert!(tags.contains(&'n') && tags.contains(&'v'), "{tags:?}"); + } + + // (b) tags that fold into one state are summed; overflow is an error + #[test] + fn folded_tags_sum_and_overflow_is_refused() { + // n 60 + p 50 = Noun 110 beats v 100; either alone would lose. + let r = [reading(b'n', Some(60)), reading(b'p', Some(50)), reading(b'v', Some(100))]; + assert_eq!(counted_pos(&r).unwrap(), Some(Pos::Noun)); + let big = [reading(b'n', Some(u64::MAX)), reading(b'p', Some(1))]; + assert_eq!(counted_pos(&big), Err(EvidenceError::CountOverflow)); + } + + // (c) one unknown count makes that whole state unknown + #[test] + fn an_unknown_count_takes_its_state_out() { + let r = [reading(b'n', Some(500)), reading(b'n', None), reading(b'v', Some(1))]; + assert_eq!(counted_pos(&r).unwrap(), Some(Pos::Verb)); + } + + // (d) `changes`: first row is the verb, the counts say noun + #[test] + fn changes_is_a_noun_by_count_and_a_verb_by_first_row() { + let forms = committed("word_forms.csv"); + let v = vocab(&["changes"]); + let t = Tagger::load("rank,lemma,PoS\n", &forms, &v).unwrap(); + assert!(!t.lemmas.contains_key("changes"), "must not be a lemma-table key"); + assert_eq!(t.pos("changes", v.id("changes").unwrap()).unwrap(), Pos::Noun); + // Anti-vacuity: the old first-wins rule tags it Verb, so the line + // above distinguishes the two rules. + assert_eq!( + load_pos_first_wins("rank,lemma,PoS\n", &forms).get("changes"), + Some(&Pos::Verb) + ); + } + + // (e) the tie order is fixed + #[test] + fn ties_go_to_the_earlier_state() { + let nv = [reading(b'v', Some(10)), reading(b'n', Some(10))]; + assert_eq!(counted_pos(&nv).unwrap(), Some(Pos::Noun)); + let vj = [reading(b'j', Some(10)), reading(b'v', Some(10))]; + assert_eq!(counted_pos(&vj).unwrap(), Some(Pos::Verb)); + } + + // (f) no known count gives None, and the tagger falls through + #[test] + fn no_known_count_falls_through_to_archaic_then_other() { + assert_eq!(counted_pos(&[reading(b'n', None)]).unwrap(), None); + assert_eq!(counted_pos(&[]).unwrap(), None); + let forms = "lemRank,lemma,PoS,lemFreq,wordFreq,word\n1,hath,n,5,,hath\n2,zz,n,5,,zz\n"; + let v = vocab(&["hath", "zz"]); + let t = Tagger::load("rank,lemma,PoS\n", forms, &v).unwrap(); + assert_eq!(t.pos("hath", v.id("hath").unwrap()).unwrap(), Pos::Verb); + assert_eq!(t.pos("zz", v.id("zz").unwrap()).unwrap(), Pos::Other); + } + + // (g) stay-silent: one reading keeps its tag + #[test] + fn a_single_reading_keeps_its_tag() { + assert_eq!(counted_pos(&[reading(b'j', Some(3))]).unwrap(), Some(Pos::Adj)); + assert_eq!(counted_pos(&[reading(b'r', Some(3))]).unwrap(), Some(Pos::Other)); + } + + // (h) + G7: the lemma table is never overruled by the forms layer + #[test] + fn the_forms_layer_never_retags_a_lemma_table_word() { + let v = vocab(&["work"]); + let t = Tagger::load(&committed("lemmas_5k.csv"), &committed("word_forms.csv"), &v).unwrap(); + let id = v.id("work").unwrap(); + // The counts alone would say Noun (n 456,169 vs v 356,692) ... + assert_eq!(counted_pos(t.evidence.readings(id)).unwrap(), Some(Pos::Noun)); + // ... and the lemma table still wins. + assert_eq!(t.pos("work", id).unwrap(), Pos::Verb); + + // The same property on a conflicting row, with the row proven loaded. + let conflicting = "lemRank,lemma,PoS,lemFreq,wordFreq,word\n1,x,n,9,4,create\n"; + let v = vocab(&["create"]); + let t = Tagger::load("rank,lemma,PoS\n1,create,v\n", conflicting, &v).unwrap(); + assert_eq!(t.evidence.reading_count(), 1); + assert_eq!(t.pos("create", v.id("create").unwrap()).unwrap(), Pos::Verb); + } + + // (i) a capitalised COCA surface still routes + #[test] + fn a_capitalised_surface_routes_after_lowercasing() { + let forms = "lemRank,lemma,PoS,lemFreq,wordFreq,word\n7,true,j,9,4,True\n"; + let v = vocab(&["true"]); + let t = Tagger::load("rank,lemma,PoS\n", forms, &v).unwrap(); + assert_eq!(t.report.unrouted, 0); + assert_eq!(t.pos("true", v.id("true").unwrap()).unwrap(), Pos::Adj); + // Without the lowercasing the same row is not routed. + let (_, raw) = load_word_forms_csv(forms, &v).unwrap(); + assert_eq!(raw.unrouted, 1); + } +} From 75119bafa953476c1c4ce4401c4d9bf0bc04b9bf Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 21:43:57 +0000 Subject: [PATCH 02/11] deepnsm-v2 bible_wave: rustfmt --- crates/deepnsm-v2/examples/bible_wave.rs | 61 +++++++++++++++++++----- 1 file changed, 48 insertions(+), 13 deletions(-) diff --git a/crates/deepnsm-v2/examples/bible_wave.rs b/crates/deepnsm-v2/examples/bible_wave.rs index 54d369fb2..5a482a925 100644 --- a/crates/deepnsm-v2/examples/bible_wave.rs +++ b/crates/deepnsm-v2/examples/bible_wave.rs @@ -29,8 +29,8 @@ use deepnsm_v2::{ load_cam96_codes, load_cam96_space, load_word_forms_csv, parse_to_spo, EvidenceError, - LexicalEvidence, LexicalReading, Nsm, PaletteVocab, Pos, Spo, Tagged, TemporalStream, WordId, - WordFormsReport, + LexicalEvidence, LexicalReading, Nsm, PaletteVocab, Pos, Spo, Tagged, TemporalStream, + WordFormsReport, WordId, }; use std::collections::HashMap; use std::path::PathBuf; @@ -137,7 +137,8 @@ fn main() { "/../deepnsm/word_frequency/word_forms.csv" )) .expect("word_forms.csv (sibling deepnsm crate)"); - let tagger = Tagger::load(&lemmas_csv, &forms_csv, &nsm.vocab).expect("word_forms.csv evidence"); + let tagger = + Tagger::load(&lemmas_csv, &forms_csv, &nsm.vocab).expect("word_forms.csv evidence"); let r = &tagger.report; println!( "LEXICON word_forms: {} rows, {} readings stored, {} empty surface, {} not in vocab", @@ -150,7 +151,9 @@ fn main() { let mut moved = 0usize; for id in 0..nsm.vocab.len() { let id = WordId::try_from(id).expect("vocab fits u16"); - let Some(w) = nsm.vocab.word(id) else { continue }; + let Some(w) = nsm.vocab.word(id) else { + continue; + }; let old = first_wins .get(w) .copied() @@ -1117,7 +1120,11 @@ struct Tagger { } impl Tagger { - fn load(lemmas_csv: &str, forms_csv: &str, vocab: &PaletteVocab) -> Result { + fn load( + lemmas_csv: &str, + forms_csv: &str, + vocab: &PaletteVocab, + ) -> Result { let mut lemmas = HashMap::new(); for line in lemmas_csv.lines().skip(1) { let f: Vec<&str> = line.split(',').collect(); @@ -1220,7 +1227,11 @@ mod tests { #[test] fn folded_tags_sum_and_overflow_is_refused() { // n 60 + p 50 = Noun 110 beats v 100; either alone would lose. - let r = [reading(b'n', Some(60)), reading(b'p', Some(50)), reading(b'v', Some(100))]; + let r = [ + reading(b'n', Some(60)), + reading(b'p', Some(50)), + reading(b'v', Some(100)), + ]; assert_eq!(counted_pos(&r).unwrap(), Some(Pos::Noun)); let big = [reading(b'n', Some(u64::MAX)), reading(b'p', Some(1))]; assert_eq!(counted_pos(&big), Err(EvidenceError::CountOverflow)); @@ -1229,7 +1240,11 @@ mod tests { // (c) one unknown count makes that whole state unknown #[test] fn an_unknown_count_takes_its_state_out() { - let r = [reading(b'n', Some(500)), reading(b'n', None), reading(b'v', Some(1))]; + let r = [ + reading(b'n', Some(500)), + reading(b'n', None), + reading(b'v', Some(1)), + ]; assert_eq!(counted_pos(&r).unwrap(), Some(Pos::Verb)); } @@ -1239,8 +1254,14 @@ mod tests { let forms = committed("word_forms.csv"); let v = vocab(&["changes"]); let t = Tagger::load("rank,lemma,PoS\n", &forms, &v).unwrap(); - assert!(!t.lemmas.contains_key("changes"), "must not be a lemma-table key"); - assert_eq!(t.pos("changes", v.id("changes").unwrap()).unwrap(), Pos::Noun); + assert!( + !t.lemmas.contains_key("changes"), + "must not be a lemma-table key" + ); + assert_eq!( + t.pos("changes", v.id("changes").unwrap()).unwrap(), + Pos::Noun + ); // Anti-vacuity: the old first-wins rule tags it Verb, so the line // above distinguishes the two rules. assert_eq!( @@ -1273,18 +1294,32 @@ mod tests { // (g) stay-silent: one reading keeps its tag #[test] fn a_single_reading_keeps_its_tag() { - assert_eq!(counted_pos(&[reading(b'j', Some(3))]).unwrap(), Some(Pos::Adj)); - assert_eq!(counted_pos(&[reading(b'r', Some(3))]).unwrap(), Some(Pos::Other)); + assert_eq!( + counted_pos(&[reading(b'j', Some(3))]).unwrap(), + Some(Pos::Adj) + ); + assert_eq!( + counted_pos(&[reading(b'r', Some(3))]).unwrap(), + Some(Pos::Other) + ); } // (h) + G7: the lemma table is never overruled by the forms layer #[test] fn the_forms_layer_never_retags_a_lemma_table_word() { let v = vocab(&["work"]); - let t = Tagger::load(&committed("lemmas_5k.csv"), &committed("word_forms.csv"), &v).unwrap(); + let t = Tagger::load( + &committed("lemmas_5k.csv"), + &committed("word_forms.csv"), + &v, + ) + .unwrap(); let id = v.id("work").unwrap(); // The counts alone would say Noun (n 456,169 vs v 356,692) ... - assert_eq!(counted_pos(t.evidence.readings(id)).unwrap(), Some(Pos::Noun)); + assert_eq!( + counted_pos(t.evidence.readings(id)).unwrap(), + Some(Pos::Noun) + ); // ... and the lemma table still wins. assert_eq!(t.pos("work", id).unwrap(), Pos::Verb); From b6f5e6695139db2b66ce95cc21e0a7cd8e8cae5e Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 21:46:52 +0000 Subject: [PATCH 03/11] board: D-LXC-1 results (KJV before/after, disable runs, 25 moved tags) --- .claude/board/AGENT_LOG.md | 13 +++ ...9-29-deepnsm-v2-counted-pick-tag-deltas.md | 41 +++++---- ...deepnsm-v2-lexical-evidence-consumer-v1.md | 89 +++++++++++++++---- 3 files changed, 108 insertions(+), 35 deletions(-) diff --git a/.claude/board/AGENT_LOG.md b/.claude/board/AGENT_LOG.md index 7f3465523..97953f281 100644 --- a/.claude/board/AGENT_LOG.md +++ b/.claude/board/AGENT_LOG.md @@ -1,3 +1,16 @@ +## 2026-09-29 — D-LXC-1 implemented (orchestrator, no agents) + +- `crates/deepnsm-v2/examples/bible_wave.rs`: counted PoS pick replaces the + `word_forms.csv` first-wins; lemma table still first. `Cargo.toml`: + `[[example]] bible_wave test = true`. 9 new tests; 124 lib tests unchanged. +- Disable runs: 8 guards, each turned its named test red. One first attempt at + the first-wins disable was not the old rule and stayed green; replaced. +- clippy `-D warnings` and fmt are clean. +- KJV before/after: 70,393 → 70,396 triples; 25 tags moved (G6 exact); 107 of + 771,176 tokens changed; long-range shares unchanged. Counts-first + alternative: 71,088 triples, 141 tags. +- `.claude/settings.json` gains `attribution` (D-LXC-7). + ## 2026-09-29 — D-LXC-1 plan rewritten with Read (orchestrator, no agents, no code) - Operator-directed. The council run below read source with shell diff --git a/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md b/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md index 809f32a39..213adde98 100644 --- a/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md +++ b/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md @@ -1,32 +1,37 @@ -# DeepNSM-v2 counted PoS pick: tag-change counts (2026-09-29) +# DeepNSM-v2 counted PoS pick: tag changes and the KJV run (2026-09-29) -**Status:** MEASURED (offline, over committed CSVs plus the released KJV -vocabulary). No code changed. +**Status:** MEASURED. D-LXC-1 is implemented in `bible_wave` (this PR). The removed `deepnsm-v2/src/lexicon.rs` (`68955ecb^`) assumed "the first row [of a frequency-ranked table] is the dominant reading". For `word_forms.csv` that is false: the file is ordered by lemma rank, and 259 surfaces have a -first row that is not their most frequent reading (e.g. `changes`: first row -v 13,624 at `:707`, noun n 113,085 at `:853`). +first row that is not their most frequent reading. For example, `changes` has +its verb row first (v 13,624 at `:707`), but the noun row (`:853`) counts +113,085. -Replacing that first-wins layer in `bible_wave::load_pos` with a counted pick -(highest summed `wordFreq` per folded parser state, tie order Noun > Verb > -Adj > Det > Other) changes: +D-LXC-1 replaces that first-wins layer with a counted pick: the highest summed +`wordFreq` per folded parser state, ties in the order Noun > Verb > Adj > Det +> Other. The lemma table stays first. -| variant | over both tables | within `bible_vocab.txt` (12,543) | -|---|---|---| -| B: lemma table first, as pinned by `ec50f07b` (the plan's choice) | 105 | **25** | -| A: counted pick first, lemma table as fallback | 288 (183 lemma-derived) | 141 | +| | before (first-wins) | after (B, shipped) | A (counts first) | +|---|---|---|---| +| in-vocabulary tags moved | — | 25 | 141 | +| KJV triples (31,102 verses) | 70,393 | 70,396 | 71,088 | +| distinct subjects | 1,227 | 1,237 | 1,243 | +| same-subject links beyond ±5 / ±8 | 60.3% / 52.7% | 60.3% / 52.7% | 60.3% / 52.6% | + +Under B, 107 of 771,176 in-vocabulary KJV tokens change tag. The offline +predictions (25 and 141 words) match the in-code G6 count exactly. Whether the +moved tags are more correct is not measured. Method: Python over `crates/deepnsm/word_frequency/{lemmas_5k,word_forms}.csv` -and `bible_vocab.txt` from release `v0.1.0-cam96-data`. Surfaces are -lowercased as `load_pos` does. Fold: `n|p→Noun, v→Verb, j→Adj, a|d→Det, -else Other`. Whether either variant tags the KJV better is not measured; the -KJV before/after is gate G5 of D-LXC-1. +and `bible_vocab.txt` (release `v0.1.0-cam96-data`); `bible_wave` run on +Gutenberg #10 with the same release artifacts, `main` `5282dfa3` against this +branch. The KJV text is input only and is not committed. Pre-existing observations filed with the plan: - `archaic_pos` never fires for COCA-known words such as `art` (D-LXC-9). -- No in-crate tagger in deepnsm-v2 produces `Pos::Rel`; external callers can - still pass it (D-LXC-10). +- No in-crate tagger produces `Pos::Rel`; external callers can still pass it + (D-LXC-10). Plan: `.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md`. diff --git a/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md b/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md index 221b06913..ef10ca542 100644 --- a/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md +++ b/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md @@ -152,29 +152,84 @@ Other facts read for this plan: ## Checklist (D-LXC-1; D-LXC-7 rides with it) -- [ ] Tests first (G3), each seen red before the change. -- [ ] In `bible_wave` only: build `LexicalEvidence` with `load_word_forms_csv` +- [x] Tests first (G3). Each was shown red by a disable run (see Results). +- [x] In `bible_wave` only: build `LexicalEvidence` with `load_word_forms_csv` against `nsm.vocab`, after lowercasing the `word` column. -- [ ] Per `WordId`: read `readings`, fold each `PosCode` with the example's +- [x] Per `WordId`: read `readings`, fold each `PosCode` with the example's `coca_pos`, sum known `form_count` per state with `checked_add` (a state with any unknown count is unknown), take the highest known sum with the fixed tie order; no known count gives `None`. -- [ ] Resolution order, variant B (F9): lemma table → counted pick → +- [x] Resolution order, variant B (F9): lemma table → counted pick → `archaic_pos` → `Pos::Other`. The lemma layer is unchanged. -- [ ] Give the pick a test home that CI runs: `Cargo.toml` has no - `[[example]]` section today, so either add one for `bible_wave` with - `test = true`, or include the functions in a `tests/` file via - `#[path]`. G1 decides which. -- [ ] Measure on the KJV (Gutenberg #10 `pg10.txt`, input only, not +- [x] Give the pick a test home that CI runs: `[[example]] name = "bible_wave"` + with `test = true`. +- [x] Measure on the KJV (Gutenberg #10 `pg10.txt`, input only, not committed): before and after for B, and A as a reported alternative. - Triples, subjects, same-subject links, % beyond ±5 and ±8, tokens whose - `Pos` changed, and `WordFormsReport` (rows, stored, empty_surface, - unrouted). -- [ ] Record the tag-fold duplication: two copies of `coca_pos` remain - (`bible_wave.rs:980`, `genre_shapes.rs:204`) under F7; a change to one - is a change to both. -- [ ] `.claude/settings.json` `"attribution": {"commit": "", "pr": ""}`. -- [ ] Not touched: `lexical.rs`, `lib.rs`, `fsm.rs`, `genre_shapes.rs`. +- [x] Record the tag-fold duplication: two copies of `coca_pos` remain + (`bible_wave.rs`, `genre_shapes.rs`) under F7; a change to one is a change + to both. +- [x] `.claude/settings.json` `"attribution": {"commit": "", "pr": ""}`. +- [x] Not touched: `lexical.rs`, `lib.rs`, `fsm.rs`, `genre_shapes.rs`. + +## Results (2026-09-29) + +**Code.** `crates/deepnsm-v2/examples/bible_wave.rs`: +- `counted_pos` implements the fold and pick. +- `Tagger` is lemma table → counted pick → `archaic_pos` → Other. +- `lowercase_word_column` fixes the capitalised COCA surfaces. +- `load_pos_first_wins` is the old rule, kept only for G6. +- Nine tests. + +`crates/deepnsm-v2/Cargo.toml` adds `[[example]] name = "bible_wave"` +`test = true`. + +**G1.** `cargo test --manifest-path crates/deepnsm-v2/Cargo.toml` runs 124 lib +tests plus the 9 new example tests; all pass. Each disable run below turned at +least one of them red under the example's tests: + +| disable | red | +|---|---| +| first reading wins (the old rule) | 6 tests, including (d) | +| no sum across folded tags | (b) | +| unknown count treated as zero | (c) and (f) | +| ties go to the later state | (e) | +| no known count becomes Noun | (f) | +| counted pick before the lemma table | (h)/G7 | +| no lowercasing | (i) | +| overflow wraps | (b) | + +A first attempt at the first-wins disable (take the first state in tie order) +stayed green. It was not the old rule, so it was replaced. + +**G2.** clippy `--all-targets -D warnings` and `fmt --check` are clean. + +**G4.** The #1299 invariance tests are unchanged and green. + +**G5/G6: KJV.** The `v0.1.0-cam96-data` release artifacts were used. Before is +`main` `5282dfa3`; after is this branch. Every in-code gate passes in all three +runs. + +| | before (first-wins) | **after (B)** | A (counts first) | +|---|---|---|---| +| verses | 31,102 | 31,102 | 31,102 | +| triples | 70,393 | **70,396** | 71,088 | +| distinct subjects | 1,227 | **1,237** | 1,243 | +| distinct predicates | 1,941 | **1,931** | 1,944 | +| same-subject links | 60,947 | **60,944** | 61,554 | +| beyond ±5 / ±8 | 60.3% / 52.7% | **60.3% / 52.7%** | 60.3% / 52.6% | +| in-vocab tags moved (G6) | — | **25** | 141 | + +- **Token scope.** Of 771,176 in-vocabulary tokens, 107 change tag under B. + The largest are bases 16, locks 15, promises 13 and flies 10. +- **`WordFormsReport`:** 11,460 rows, 4,284 readings stored, 4 empty + surface, 7,172 not in the KJV vocabulary. +- **Reading.** B is a small correction. It moves 25 words' tags, adds 3 + triples, and leaves the long-range shares unchanged. Whether the moved tags + are more correct is still not measured. A moves 5.6× more tags and adds 695 + triples. +- **G7.** The lemma-first property is pinned again by + `the_forms_layer_never_retags_a_lemma_table_word`. That is the falsifier + `TD-DEEPNSM-V2-SESSION-RESIDUE` item 3 asked for. ## Pre-registered gates From 25ed5021d1967b9461f5d3ed907975cf090248dc Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 21:53:45 +0000 Subject: [PATCH 04/11] deepnsm-v2: frequency-ordered register with percentile coverage (D-LXC-1) LexicalEvidence now stores each word's readings most frequent first (unknown counts last, ties in file order) and exposes coverage(id): the cumulative percentile coverage of each reading, None when any count of the word is unknown. bible_wave reads position 0 instead of summing counts per folded state at tag time; counted_pos and PICK_ORDER are gone. KJV is unchanged from the summed pick: 25 moved tags, 70,396 triples. Claude-Session: https://claude.ai/code/session_01AUbvJZf4pD6GzkZZ6LtqSE --- ...9-29-deepnsm-v2-counted-pick-tag-deltas.md | 5 + ...deepnsm-v2-lexical-evidence-consumer-v1.md | 17 ++ crates/deepnsm-v2/examples/bible_wave.rs | 197 +++++++----------- crates/deepnsm-v2/src/lexical.rs | 133 +++++++++++- 4 files changed, 220 insertions(+), 132 deletions(-) diff --git a/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md b/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md index 213adde98..9c926234a 100644 --- a/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md +++ b/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md @@ -34,4 +34,9 @@ Pre-existing observations filed with the plan: - No in-crate tagger produces `Pos::Rel`; external callers can still pass it (D-LXC-10). +**Amended (same day).** The summed pick was replaced by the register: +`LexicalEvidence` stores readings most frequent first with a cumulative +percentile coverage, and the tagger reads position 0. The KJV run is identical +to "after (B)" above (25 moved, 70,396 triples); the summing moved nothing. + Plan: `.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md`. diff --git a/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md b/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md index ef10ca542..58a676c71 100644 --- a/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md +++ b/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md @@ -231,6 +231,23 @@ runs. `the_forms_layer_never_retags_a_lemma_table_word`. That is the falsifier `TD-DEEPNSM-V2-SESSION-RESIDUE` item 3 asked for. +### Amendment: frequency is the register (2026-09-29, operator-directed) + +The counted pick summed counts at tag time. Frequency is a dimension, so it +belongs in the storage order, expressed as percentile coverage: +- `LexicalEvidenceBuilder::finish` stores each word's readings most frequent + first (unknown last, ties in file order). `LexicalEvidence::coverage(id)` + gives each reading's cumulative percentile coverage (integer `0..=100`, + `None` when any count of the word is unknown). `changes`: noun first, 89. +- `bible_wave`'s `dominant_pos` reads position 0 when coverage is known. + `counted_pos`, `PICK_ORDER` and per-state summing are removed. +- Behaviour dropped with the summing: folded tags no longer add (n 60 + p 50 + loses to v 100), and ties keep file order instead of Noun > Verb. +- KJV: identical to the summed run — 25 moved tags, 70,396 triples, same + shares. The summing moved nothing on this corpus. +- This supersedes the "no library change" non-goal (F1): the change is storage + order plus a derived integer, not interpretation. + ## Pre-registered gates - G1 `cargo test --manifest-path crates/deepnsm-v2/Cargo.toml` green; a diff --git a/crates/deepnsm-v2/examples/bible_wave.rs b/crates/deepnsm-v2/examples/bible_wave.rs index 5a482a925..96046129a 100644 --- a/crates/deepnsm-v2/examples/bible_wave.rs +++ b/crates/deepnsm-v2/examples/bible_wave.rs @@ -29,8 +29,7 @@ use deepnsm_v2::{ load_cam96_codes, load_cam96_space, load_word_forms_csv, parse_to_spo, EvidenceError, - LexicalEvidence, LexicalReading, Nsm, PaletteVocab, Pos, Spo, Tagged, TemporalStream, - WordFormsReport, WordId, + LexicalEvidence, Nsm, PaletteVocab, Pos, Spo, Tagged, TemporalStream, WordFormsReport, WordId, }; use std::collections::HashMap; use std::path::PathBuf; @@ -144,7 +143,7 @@ fn main() { "LEXICON word_forms: {} rows, {} readings stored, {} empty surface, {} not in vocab", r.rows, r.stored, r.empty_surface, r.unrouted ); - // G6 (D-LXC-1) — how many in-vocabulary tags the counted pick moves away + // G6 (D-LXC-1) — how many in-vocabulary tags the register pick moves away // from the old first-wins rule. Pinned against the released // `bible_vocab.txt`: any other number means the tables or the rule changed. let first_wins = load_pos_first_wins(&lemmas_csv, &forms_csv); @@ -159,15 +158,15 @@ fn main() { .copied() .or_else(|| archaic_pos(w)) .unwrap_or(Pos::Other); - if tagger.pos(w, id).expect("count sum") != old { + if tagger.pos(w, id) != old { moved += 1; } } assert_eq!( moved, 25, - "KILL G6: the counted pick moved {moved} in-vocabulary tags, pinned 25" + "KILL G6: the register pick moved {moved} in-vocabulary tags, pinned 25" ); - println!("G6 PASS counted pick moves {moved} in-vocabulary tags from first-wins (pinned 25)"); + println!("G6 PASS register pick moves {moved} in-vocabulary tags from first-wins (pinned 25)"); // ── stream: verse index = version; FSM → SPO ── let mut stream = TemporalStream::new(); @@ -178,7 +177,7 @@ fn main() { for tok in verse.split_whitespace() { let Some(w) = normalise(tok) else { continue }; let Some(id) = nsm.vocab.id(&w) else { continue }; - let pos = tagger.pos(&w, id).expect("count sum"); + let pos = tagger.pos(&w, id); tagged_buf.push(Tagged::new(id, pos)); } tagged_buf.push(Tagged::new(0, Pos::Stop)); // verse boundary flushes @@ -1045,45 +1044,21 @@ fn normalise(tok: &str) -> Option { (w.len() >= 2).then_some(w) } -/// The fold's states in tie-break order: a tie between two known sums goes to -/// the state listed first. -const PICK_ORDER: [Pos; 5] = [Pos::Noun, Pos::Verb, Pos::Adj, Pos::Det, Pos::Other]; - -/// The counted reading of one word (D-LXC-1). +/// The dominant reading of one word (D-LXC-1): position 0 of the +/// frequency-ordered register, folded with [`coca_pos`]. /// -/// Folds each reading's source tag with [`coca_pos`] and sums the known -/// `wordFreq` per parser state. A state with any unknown count is unknown and -/// does not compete; overflow is an error, never a wrap. The highest known sum -/// wins, ties go to [`PICK_ORDER`], and no known count at all is `None` — the -/// caller's fallback decides, nothing is invented here. +/// `LexicalEvidence` stores readings most frequent first, so nothing is summed +/// here. The register is read only when its coverage is known; with any +/// unknown count the dominant reading is unknown and this returns `None`, so +/// the caller's fallback decides. /// /// This replaces taking the FIRST `word_forms.csv` row: that file is ordered by /// lemma rank, not by surface frequency, so for 259 surfaces the first row is /// not the dominant reading (`changes`: verb row 13,624 first, noun 113,085). -fn counted_pos(readings: &[LexicalReading]) -> Result, EvidenceError> { - let mut best: Option<(u64, Pos)> = None; - for state in PICK_ORDER { - let mut sum = Some(0u64); - let mut seen = false; - for r in readings { - if coca_pos(&r.pos.as_char().to_string()) != state { - continue; - } - seen = true; - sum = match (sum, r.form_count) { - (Some(s), Some(c)) => Some(s.checked_add(c).ok_or(EvidenceError::CountOverflow)?), - _ => None, - }; - } - let (true, Some(sum)) = (seen, sum) else { - continue; - }; - // Strictly greater: an equal sum keeps the earlier state in PICK_ORDER. - if best.is_none_or(|(b, _)| sum > b) { - best = Some((sum, state)); - } - } - Ok(best.map(|(_, p)| p)) +fn dominant_pos(evidence: &LexicalEvidence, id: WordId) -> Option { + evidence.coverage(id).first().copied().flatten()?; + let r = evidence.readings(id).first()?; + Some(coca_pos(&r.pos.as_char().to_string())) } /// Lowercase the `word` column of `word_forms.csv`, leaving the header and the @@ -1108,7 +1083,7 @@ fn lowercase_word_column(forms_csv: &str) -> String { out } -/// The corpus tagger: lemma table → counted `word_forms.csv` evidence → +/// The corpus tagger: lemma table → dominant `word_forms.csv` reading → /// [`archaic_pos`] → [`Pos::Other`]. /// /// Lemma-first is deliberate and pinned (commit `ec50f07b`): the forms layer @@ -1144,14 +1119,13 @@ impl Tagger { } /// The tag for word `w` whose routing id is `id`. - fn pos(&self, w: &str, id: WordId) -> Result { + fn pos(&self, w: &str, id: WordId) -> Pos { if let Some(&p) = self.lemmas.get(w) { - return Ok(p); - } - if let Some(p) = counted_pos(self.evidence.readings(id))? { - return Ok(p); + return p; } - Ok(archaic_pos(w).unwrap_or(Pos::Other)) + dominant_pos(&self.evidence, id) + .or_else(|| archaic_pos(w)) + .unwrap_or(Pos::Other) } } @@ -1185,15 +1159,6 @@ fn load_pos_first_wins(lemmas_csv: &str, forms_csv: &str) -> HashMap) -> LexicalReading { - LexicalReading { - pos: PosCode(tag), - lemma: None, - form_count: count, - } - } fn vocab(words: &[&str]) -> PaletteVocab { let mut v = PaletteVocab::new(); @@ -1209,11 +1174,23 @@ mod tests { .expect("committed COCA table") } + const NO_LEMMAS: &str = "rank,lemma,PoS\n"; + + fn forms(rows: &str) -> String { + format!("lemRank,lemma,PoS,lemFreq,wordFreq,word\n{rows}") + } + + fn tag(forms_csv: &str, w: &str) -> Pos { + let v = vocab(&[w]); + let t = Tagger::load(NO_LEMMAS, forms_csv, &v).unwrap(); + t.pos(w, v.id(w).unwrap()) + } + // (a) a homograph keeps every reading #[test] fn record_keeps_its_noun_and_verb_readings() { let v = vocab(&["record"]); - let t = Tagger::load("rank,lemma,PoS\n", &committed("word_forms.csv"), &v).unwrap(); + let t = Tagger::load(NO_LEMMAS, &committed("word_forms.csv"), &v).unwrap(); let tags: Vec = t .evidence .readings(v.id("record").unwrap()) @@ -1223,85 +1200,64 @@ mod tests { assert!(tags.contains(&'n') && tags.contains(&'v'), "{tags:?}"); } - // (b) tags that fold into one state are summed; overflow is an error + // (b) the pick is position 0 of the register: folded tags are NOT summed #[test] - fn folded_tags_sum_and_overflow_is_refused() { - // n 60 + p 50 = Noun 110 beats v 100; either alone would lose. - let r = [ - reading(b'n', Some(60)), - reading(b'p', Some(50)), - reading(b'v', Some(100)), - ]; - assert_eq!(counted_pos(&r).unwrap(), Some(Pos::Noun)); - let big = [reading(b'n', Some(u64::MAX)), reading(b'p', Some(1))]; - assert_eq!(counted_pos(&big), Err(EvidenceError::CountOverflow)); + fn the_dominant_reading_wins_without_summing() { + // n 60 + p 50 would be Noun 110 if summed; the register says v 100. + assert_eq!( + tag(&forms("1,x,n,9,60,w\n2,y,p,9,50,w\n3,z,v,9,100,w\n"), "w"), + Pos::Verb + ); } - // (c) one unknown count makes that whole state unknown + // (c) an unknown count makes the register unreadable, so the fallback decides #[test] - fn an_unknown_count_takes_its_state_out() { - let r = [ - reading(b'n', Some(500)), - reading(b'n', None), - reading(b'v', Some(1)), - ]; - assert_eq!(counted_pos(&r).unwrap(), Some(Pos::Verb)); + fn an_unknown_count_falls_through() { + assert_eq!(tag(&forms("1,x,n,9,500,w\n2,y,v,9,,w\n"), "w"), Pos::Other); } - // (d) `changes`: first row is the verb, the counts say noun + // (d) `changes`: first row is the verb, the register says noun (89%) #[test] - fn changes_is_a_noun_by_count_and_a_verb_by_first_row() { + fn changes_is_a_noun_by_frequency_and_a_verb_by_first_row() { let forms = committed("word_forms.csv"); let v = vocab(&["changes"]); - let t = Tagger::load("rank,lemma,PoS\n", &forms, &v).unwrap(); + let t = Tagger::load(NO_LEMMAS, &forms, &v).unwrap(); + let id = v.id("changes").unwrap(); assert!( !t.lemmas.contains_key("changes"), "must not be a lemma-table key" ); + assert_eq!(t.pos("changes", id), Pos::Noun); + assert_eq!(t.evidence.coverage(id).first().copied().flatten(), Some(89)); + // Anti-vacuity: the old first-wins rule tags it Verb. assert_eq!( - t.pos("changes", v.id("changes").unwrap()).unwrap(), - Pos::Noun - ); - // Anti-vacuity: the old first-wins rule tags it Verb, so the line - // above distinguishes the two rules. - assert_eq!( - load_pos_first_wins("rank,lemma,PoS\n", &forms).get("changes"), + load_pos_first_wins(NO_LEMMAS, &forms).get("changes"), Some(&Pos::Verb) ); } - // (e) the tie order is fixed + // (e) equal counts keep file order #[test] - fn ties_go_to_the_earlier_state() { - let nv = [reading(b'v', Some(10)), reading(b'n', Some(10))]; - assert_eq!(counted_pos(&nv).unwrap(), Some(Pos::Noun)); - let vj = [reading(b'j', Some(10)), reading(b'v', Some(10))]; - assert_eq!(counted_pos(&vj).unwrap(), Some(Pos::Verb)); + fn ties_keep_file_order() { + assert_eq!(tag(&forms("1,x,v,9,10,w\n2,y,n,9,10,w\n"), "w"), Pos::Verb); + assert_eq!(tag(&forms("1,x,n,9,10,w\n2,y,v,9,10,w\n"), "w"), Pos::Noun); } - // (f) no known count gives None, and the tagger falls through + // (f) no known count falls through to archaic, then Other #[test] fn no_known_count_falls_through_to_archaic_then_other() { - assert_eq!(counted_pos(&[reading(b'n', None)]).unwrap(), None); - assert_eq!(counted_pos(&[]).unwrap(), None); - let forms = "lemRank,lemma,PoS,lemFreq,wordFreq,word\n1,hath,n,5,,hath\n2,zz,n,5,,zz\n"; + let f = forms("1,hath,n,5,,hath\n2,zz,n,5,,zz\n"); let v = vocab(&["hath", "zz"]); - let t = Tagger::load("rank,lemma,PoS\n", forms, &v).unwrap(); - assert_eq!(t.pos("hath", v.id("hath").unwrap()).unwrap(), Pos::Verb); - assert_eq!(t.pos("zz", v.id("zz").unwrap()).unwrap(), Pos::Other); + let t = Tagger::load(NO_LEMMAS, &f, &v).unwrap(); + assert_eq!(t.pos("hath", v.id("hath").unwrap()), Pos::Verb); + assert_eq!(t.pos("zz", v.id("zz").unwrap()), Pos::Other); } - // (g) stay-silent: one reading keeps its tag + // (g) stay-silent: one reading keeps its tag and covers 100% #[test] fn a_single_reading_keeps_its_tag() { - assert_eq!( - counted_pos(&[reading(b'j', Some(3))]).unwrap(), - Some(Pos::Adj) - ); - assert_eq!( - counted_pos(&[reading(b'r', Some(3))]).unwrap(), - Some(Pos::Other) - ); + assert_eq!(tag(&forms("1,x,j,9,3,w\n"), "w"), Pos::Adj); + assert_eq!(tag(&forms("1,x,r,9,3,w\n"), "w"), Pos::Other); } // (h) + G7: the lemma table is never overruled by the forms layer @@ -1315,32 +1271,29 @@ mod tests { ) .unwrap(); let id = v.id("work").unwrap(); - // The counts alone would say Noun (n 456,169 vs v 356,692) ... - assert_eq!( - counted_pos(t.evidence.readings(id)).unwrap(), - Some(Pos::Noun) - ); + // The register alone says Noun ... + assert_eq!(dominant_pos(&t.evidence, id), Some(Pos::Noun)); // ... and the lemma table still wins. - assert_eq!(t.pos("work", id).unwrap(), Pos::Verb); + assert_eq!(t.pos("work", id), Pos::Verb); // The same property on a conflicting row, with the row proven loaded. - let conflicting = "lemRank,lemma,PoS,lemFreq,wordFreq,word\n1,x,n,9,4,create\n"; + let conflicting = forms("1,x,n,9,4,create\n"); let v = vocab(&["create"]); - let t = Tagger::load("rank,lemma,PoS\n1,create,v\n", conflicting, &v).unwrap(); + let t = Tagger::load("rank,lemma,PoS\n1,create,v\n", &conflicting, &v).unwrap(); assert_eq!(t.evidence.reading_count(), 1); - assert_eq!(t.pos("create", v.id("create").unwrap()).unwrap(), Pos::Verb); + assert_eq!(t.pos("create", v.id("create").unwrap()), Pos::Verb); } // (i) a capitalised COCA surface still routes #[test] fn a_capitalised_surface_routes_after_lowercasing() { - let forms = "lemRank,lemma,PoS,lemFreq,wordFreq,word\n7,true,j,9,4,True\n"; + let f = forms("7,true,j,9,4,True\n"); let v = vocab(&["true"]); - let t = Tagger::load("rank,lemma,PoS\n", forms, &v).unwrap(); + let t = Tagger::load(NO_LEMMAS, &f, &v).unwrap(); assert_eq!(t.report.unrouted, 0); - assert_eq!(t.pos("true", v.id("true").unwrap()).unwrap(), Pos::Adj); + assert_eq!(t.pos("true", v.id("true").unwrap()), Pos::Adj); // Without the lowercasing the same row is not routed. - let (_, raw) = load_word_forms_csv(forms, &v).unwrap(); + let (_, raw) = load_word_forms_csv(&f, &v).unwrap(); assert_eq!(raw.unrouted, 1); } } diff --git a/crates/deepnsm-v2/src/lexical.rs b/crates/deepnsm-v2/src/lexical.rs index e0d26707b..fb2a30a5f 100644 --- a/crates/deepnsm-v2/src/lexical.rs +++ b/crates/deepnsm-v2/src/lexical.rs @@ -28,8 +28,16 @@ //! `codes[word_id]` table. Semantic distance comes from Cam96; counts come //! from here; the two are never mixed. //! - **Not truth.** Counts are observed population evidence, stored as exact -//! integers. No normalisation to `f32`, no probability, no NARS truth — a -//! consumer that wants a ratio computes it from the integers it asked for. +//! integers. No normalisation to `f32`, no probability, no NARS truth. +//! +//! ## Frequency is the register's order +//! +//! A word's readings are stored most frequent first, and each carries an +//! integer cumulative percentile coverage ([`LexicalEvidence::coverage`]). The +//! dominant reading is position 0 and its share is `coverage[0]` — a reader +//! never sums counts to find it. Source file order is not kept: COCA orders +//! `word_forms.csv` by lemma rank, so its first row is not the dominant reading +//! for 259 surfaces (`changes`: verb 13,624 first, noun 113,085 second). //! //! ## Unknown is not zero //! @@ -251,10 +259,25 @@ impl LexicalEvidenceBuilder { Ok(()) } - /// Freeze into id-indexed storage. Readings of one word keep insertion order. + /// Freeze into id-indexed storage. + /// + /// Each word's readings are stored in FREQUENCY ORDER — highest known + /// `form_count` first, unknown counts last, equal counts in insertion + /// order — and each carries its cumulative percentile coverage (see + /// [`LexicalEvidence::coverage`]). Frequency is the register's order, so a + /// reader takes position 0 for the dominant reading and never re-sums. #[must_use] pub fn finish(mut self) -> LexicalEvidence { - self.readings.sort_by_key(|(id, _)| *id); + // Stable: equal keys keep insertion order. + self.readings.sort_by_key(|(id, r)| { + ( + *id, + match r.form_count { + Some(c) => (0u8, std::cmp::Reverse(c)), + None => (1u8, std::cmp::Reverse(0)), + }, + ) + }); let mut offsets = vec![0u32; self.vocab_len + 1]; for (id, _) in &self.readings { offsets[*id as usize + 1] += 1; @@ -262,6 +285,25 @@ impl LexicalEvidenceBuilder { for i in 1..offsets.len() { offsets[i] += offsets[i - 1]; } + let mut coverage = vec![None; self.readings.len()]; + for w in offsets.windows(2) { + let (a, b) = (w[0] as usize, w[1] as usize); + let counts: Option> = self.readings[a..b] + .iter() + .map(|(_, r)| r.form_count) + .collect(); + let Some(counts) = counts else { continue }; + let total: u128 = counts.iter().map(|&c| u128::from(c)).sum(); + if total == 0 { + continue; + } + let mut cum = 0u128; + for (slot, c) in coverage[a..b].iter_mut().zip(counts) { + cum += u128::from(c); + // cum <= total, so the quotient is 0..=100. + *slot = Some((cum * 100 / total) as u8); + } + } let mut lemmas_by_name: HashMap> = HashMap::new(); for (i, e) in self.lemmas.iter().enumerate() { if let Some(name) = &e.lemma { @@ -274,6 +316,7 @@ impl LexicalEvidenceBuilder { LexicalEvidence { offsets, readings: self.readings.into_iter().map(|(_, r)| r).collect(), + coverage, lemmas: self.lemmas, lemmas_by_name, } @@ -285,7 +328,10 @@ impl LexicalEvidenceBuilder { pub struct LexicalEvidence { /// `offsets[id]..offsets[id + 1]` are word `id`'s readings. offsets: Vec, + /// Frequency-ordered within each word (see [`LexicalEvidenceBuilder::finish`]). readings: Vec, + /// Parallel to `readings`: cumulative percentile coverage. + coverage: Vec>, lemmas: Vec, lemmas_by_name: HashMap>, } @@ -306,17 +352,34 @@ fn sum_known(counts: impl Iterator>) -> Result, E } impl LexicalEvidence { - /// Every counted reading of word `id` (empty if none survived, or `id` is - /// out of range). - #[must_use] - pub fn readings(&self, id: WordId) -> &[LexicalReading] { + fn span(&self, id: WordId) -> std::ops::Range { let i = id as usize; match (self.offsets.get(i), self.offsets.get(i + 1)) { - (Some(&a), Some(&b)) => &self.readings[a as usize..b as usize], - _ => &[], + (Some(&a), Some(&b)) => a as usize..b as usize, + _ => 0..0, } } + /// Every counted reading of word `id`, most frequent first (empty if none + /// survived, or `id` is out of range). Position 0 is the dominant reading + /// when [`coverage`](Self::coverage) is known. + #[must_use] + pub fn readings(&self, id: WordId) -> &[LexicalReading] { + &self.readings[self.span(id)] + } + + /// Cumulative percentile coverage, aligned with [`readings`](Self::readings): + /// entry `k` is the percent (`0..=100`, floored) of word `id`'s known + /// occurrences covered by readings `0..=k`. The last entry is `100`, and + /// entry 0 is the dominant reading's own share. + /// + /// All `None` when any reading's count is unknown or every count is zero — + /// a share of an unknown total is itself unknown. + #[must_use] + pub fn coverage(&self, id: WordId) -> &[Option] { + &self.coverage[self.span(id)] + } + /// The lemma entry behind a reading. #[must_use] pub fn lemma(&self, r: LemmaRef) -> Option<&LemmaEntry> { @@ -536,6 +599,56 @@ mod tests { assert_eq!(e.surface_count(id), Ok(Some(133_062))); } + /// The register is frequency-ordered, not file-ordered, and carries its + /// cumulative percentile coverage. The counts are COCA's for `changes` + /// (verb row first in the file), placed on the fixture surface `record`. + /// Falsified by dropping the count sort. + #[test] + fn readings_are_frequency_ordered_with_percentile_coverage() { + let v = vocab(); + let text = format!( + "{WORD_FORMS_HEADER}\n\ + 700,change,v,200000,13624,record\n\ + 850,change,n,300000,113085,record\n" + ); + let (e, _) = load_word_forms_csv(&text, &v).unwrap(); + let id = v.id("record").unwrap(); + let rs = e.readings(id); + assert_eq!(rs[0].pos, N, "the noun row counts more and must come first"); + assert_eq!(rs[0].form_count, Some(113_085)); + assert_eq!(rs[1].pos, V); + // 113,085 / 126,709 = 89.2% -> 89; the last entry covers everything. + assert_eq!(e.coverage(id), &[Some(89), Some(100)]); + } + + /// Equal counts keep file order; unknown counts sort last and make the + /// whole word's coverage unknown; an all-zero word has no coverage. + #[test] + fn ties_unknowns_and_zeros_in_the_register() { + let v = vocab(); + let text = format!( + "{WORD_FORMS_HEADER}\n\ + 1,a,v,9,10,record\n\ + 2,a,n,9,10,record\n\ + 3,b,v,9,,records\n\ + 4,b,n,9,4,records\n\ + 5,c,n,9,0,may\n" + ); + let (e, _) = load_word_forms_csv(&text, &v).unwrap(); + let rec = v.id("record").unwrap(); + assert_eq!(e.readings(rec)[0].pos, V, "tie keeps file order"); + assert_eq!(e.coverage(rec), &[Some(50), Some(100)]); + let recs = v.id("records").unwrap(); + assert_eq!( + e.readings(recs)[0].form_count, + Some(4), + "unknown sorts last" + ); + assert_eq!(e.coverage(recs), &[None, None]); + assert_eq!(e.coverage(v.id("may").unwrap()), &[None]); + assert!(e.coverage(v.id("the").unwrap()).is_empty()); + } + /// Test 2: counts are exact integers, including one no `f32` can hold. #[test] fn integer_counts_round_trip_exactly() { From a1fbc9b19f08f8b2df79ef90859a704cecc25b01 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 22:50:27 +0000 Subject: [PATCH 05/11] board: make the D-LXC-1 tag counts reproducible Name the bible_wave command whose in-code G6 gate recomputes the 25, state that the 141 has no gate, and record the vocab asset digest as provenance. Claude-Session: https://claude.ai/code/session_01AUbvJZf4pD6GzkZZ6LtqSE --- .../2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md b/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md index 9c926234a..db87f2807 100644 --- a/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md +++ b/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md @@ -29,6 +29,14 @@ and `bible_vocab.txt` (release `v0.1.0-cam96-data`); `bible_wave` run on Gutenberg #10 with the same release artifacts, `main` `5282dfa3` against this branch. The KJV text is input only and is not committed. +Reproduce: fetch the release assets per `crates/deepnsm-v2/data/README.md`, +then `DEEPNSM_V2_DATA= cargo run --release --example bible_wave -- +`. The in-code G6 count recomputes the 25 and asserts it. The 141 +(variant A) was measured once in Python and is not re-run by any gate. The +`bible_vocab.txt` used had SHA-256 +`8dc3a65dcd3af38a2f53308fb96ef5fb5b34336c14587f6971b75ac966c3e212` (recorded +as provenance, not a gate). + Pre-existing observations filed with the plan: - `archaic_pos` never fires for COCA-known words such as `art` (D-LXC-9). - No in-crate tagger produces `Pos::Rel`; external callers can still pass it From 744ded84bff3a06f91db7c18484a52dab7e838d3 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 30 Sep 2026 02:54:28 +0000 Subject: [PATCH 06/11] deepnsm-v2 bible_wave: coverage bands at population quartiles (D-LXC-11) Each word's dominant reading share is banded Contested / Leaning / Decisive at the quartiles of the band population (non-lemma words with known coverage whose readings fold to >= 2 parser states), calibrated once at load with ndarray's rank_per_10000 rule. Report only: no tag reads a band. KJV vocabulary: population 141, cuts (72, 97), bands 34/71/36, pinned by G8c and equal to the plan's receipt. Triples unchanged (70,396). Designed through a 5+3 council; plan deepnsm-v2-coverage-bands-v1. Claude-Session: https://claude.ai/code/session_01AUbvJZf4pD6GzkZZ6LtqSE --- .claude/board/AGENT_LOG.md | 16 + .../2026-09-30-deepnsm-v2-coverage-bands.md | 33 ++ .claude/board/entries/README.md | 3 +- .claude/plans/deepnsm-v2-coverage-bands-v1.md | 234 +++++++++++++++ crates/deepnsm-v2/examples/bible_wave.rs | 283 ++++++++++++++++++ 5 files changed, 568 insertions(+), 1 deletion(-) create mode 100644 .claude/board/entries/2026-09-30-deepnsm-v2-coverage-bands.md create mode 100644 .claude/plans/deepnsm-v2-coverage-bands-v1.md diff --git a/.claude/board/AGENT_LOG.md b/.claude/board/AGENT_LOG.md index 97953f281..c41460051 100644 --- a/.claude/board/AGENT_LOG.md +++ b/.claude/board/AGENT_LOG.md @@ -1,3 +1,19 @@ +## 2026-09-30 — D-LXC-11 coverage bands (5+3 council, orchestrator implements) + +- The 5 (Sonnet): prior-art, iron-rule, code truth (general-purpose, + runtime-archaeologist charter), cascade-impact, creative-explorer. + 0 VIOLATES; findings drove 17 v1→v2 changes (rank rule from ndarray + `rank_per_10000`, D-LXC-12 dropped for D-LXC-5, `CoverageBand` naming, + T3 split, reading-share key F8). +- The 3 (Sonnet): overclaim-auditor, dilution-collapse-sentinel, + firewall-warden. 0 BLOCK, 3 P1, 9 P2; v2→v3 fixed one rank rule named + everywhere, receipts for every number, gate renames G8a-c, commit contents. +- Code: `bible_wave.rs` gains `CoverageBand`, `BandCuts`, `calibrate`, G8c; + 8 new tests (T1-T8). No tag changes. +- Gates: 126 lib + 17 example tests, clippy `-D warnings`, fmt clean. KJV: + 70,396 triples, G6 = 25, G8c = population 141, cuts (72, 97), bands + 34/71/36 — equal to the Python receipt in the plan. + ## 2026-09-29 — D-LXC-1 implemented (orchestrator, no agents) - `crates/deepnsm-v2/examples/bible_wave.rs`: counted PoS pick replaces the diff --git a/.claude/board/entries/2026-09-30-deepnsm-v2-coverage-bands.md b/.claude/board/entries/2026-09-30-deepnsm-v2-coverage-bands.md new file mode 100644 index 000000000..e994f59a3 --- /dev/null +++ b/.claude/board/entries/2026-09-30-deepnsm-v2-coverage-bands.md @@ -0,0 +1,33 @@ +# DeepNSM-v2 coverage bands at population-calibrated thresholds (2026-09-30) + +**Status:** MEASURED. D-LXC-11 is implemented in `bible_wave` (report only; +no tag changes). + +Each word's dominant reading share (`coverage(id)[0]`, from D-LXC-1's +frequency-ordered register) is banded Contested / Leaning / Decisive at the +quartiles of the band population, calibrated once at load. The population is +the vocabulary words that are not lemma-table keys, have known coverage, and +have readings that fold to at least two parser states. + +On the release KJV vocabulary (`v0.1.0-cam96-data`): + +| quantity | value | +|---|---| +| band population | 141 (of 635 cross-state words; 494 are lemma-table keys) | +| cuts (lo, hi), `rank_per_10000` indices 35 / 105 | 72, 97 | +| share range | 50 … 99 | +| contested / leaning / decisive | 34 / 71 / 36 | +| absolute alternative "< 60 contested" | 20 | + +Contested words include flies, leaves, lies, promises, needs, means. + +The Rust gate G8c and the Python receipt in the plan give the same numbers. +The KJV run is otherwise unchanged (70,396 triples, G6 = 25). + +Open for the operator: relative bands (this) vs an absolute floor. + +Ready-to-paste STATUS_BOARD row (the file is deny-listed for the agent): + +`| D-LXC-11 | coverage bands (population quartiles) in bible_wave | In PR | deepnsm-v2-coverage-bands-v1 |` + +Plan: `.claude/plans/deepnsm-v2-coverage-bands-v1.md`. diff --git a/.claude/board/entries/README.md b/.claude/board/entries/README.md index 393a6e819..64fd79d33 100644 --- a/.claude/board/entries/README.md +++ b/.claude/board/entries/README.md @@ -25,10 +25,11 @@ index row, (3) no duplicate entry id. Checks 1 and 2 are deliberately opposite directions; the stranding this convention prevents shows up in exactly one of them, never both. -176 entries, 2026-08-06 .. 2026-09-29. +177 entries, 2026-08-06 .. 2026-09-30. | date | entry id | finding | file | |---|---|---|---| +| 2026-09-30 | `deepnsm-v2-coverage-bands` | | [2026-09-30-deepnsm-v2-coverage-bands.md](2026-09-30-deepnsm-v2-coverage-bands.md) | | 2026-09-29 | `deepnsm-v2-counted-pick-tag-deltas` | | [2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md](2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md) | | 2026-09-26 | `deepnsm-v2-lexical-evidence-survives-routing` | | [2026-09-26-deepnsm-v2-lexical-evidence-survives-routing.md](2026-09-26-deepnsm-v2-lexical-evidence-survives-routing.md) | | 2026-09-25 | `window-scheduling-and-two-level-ternlog` | | [2026-09-25-window-scheduling-and-two-level-ternlog.md](2026-09-25-window-scheduling-and-two-level-ternlog.md) | diff --git a/.claude/plans/deepnsm-v2-coverage-bands-v1.md b/.claude/plans/deepnsm-v2-coverage-bands-v1.md new file mode 100644 index 000000000..07342b652 --- /dev/null +++ b/.claude/plans/deepnsm-v2-coverage-bands-v1.md @@ -0,0 +1,234 @@ +# DeepNSM-v2 coverage bands at population-calibrated thresholds (D-LXC-11) + +**Status:** RATIFIED v3 (5+3 council, 2026-09-30). Implementation in this PR. +**Parent:** `.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md` (D-LXC-1, +the frequency-ordered register this reads). + +## Overview + +D-LXC-1 stores each word's readings most frequent first, with a cumulative +percentile coverage per reading. The operator asked that frequency be read as +percentile coverage with thresholds at prevalence boundaries: + +> "It should simply be percentile coverage" +> "Compared with HDR popcount stacking early exit Belichtungsmesser statistical +> confidence interval thresholds preheating rolling floor bucket assignment. +> It's thresholds at certain Prävalenz boundaries/percentile coverage. +> Akin to using Gini palette when looking at poor countries. Or IQ buckets in +> gaussian distribution." + +This plan turns the dominant reading's share into a three-level +`CoverageBand` whose cut points are quartiles of the measured population, so +each band holds a known share of it. The cuts are calibrated once at load from +the evidence present. That is a load-time snapshot, not a streaming rolling +floor; the streaming form exists as ndarray `hpc::rolling_floor::RollingFloor`. + +## Checklist + +- [x] Spec v1, 5 research savants, draft v2, 3 reviewers, v3 (this file) +- [x] `CoverageBand`, `BandCuts`, `calibrate` in `bible_wave.rs` +- [ ] Tests T1-T8, each disable-verified red before landing +- [x] Gates G8a-G8c +- [x] Board entry + indexes in the same commit + +## Frozen decisions + +- F1 `LexicalEvidence` is evidence, not interpretation. It stores readings + frequency-first with cumulative coverage (`lexical.rs` `finish()` and + `coverage()`, PR #1304). It gets no band logic. +- F2 The PoS fold (`coca_pos`) and all tagging live in the consumer example + `crates/deepnsm-v2/examples/bible_wave.rs` (commit `68955ecb`). +- F3 Lemma table first (commit `ec50f07b`, pinned by + `the_forms_layer_never_retags_a_lemma_table_word`). The register is read only + for words the lemma table does not know. +- F4 DeepNSM-v2 is the inbound leg: no reasoning, NARS truth or belief in it. +- F5 Unknown is not zero: a word with unknown coverage has no band. +- F6 No model identifiers or advertising; Read for source inspection; the + board files denied in `.claude/settings.json` are not edited by the agent. +- F7 Search is navigation, never evidence. +- F8 The band key is the dominant READING's share, `coverage(id)[0]`. The + operator rejected summing at tag time ("why are you counting … it should + simply be percentile coverage"), so no parser-state sum is formed. +- F9 Bands are relative by design: the operator asked for cuts at prevalence + boundaries ("IQ buckets", "Gini palette"), i.e. cuts that each hold a known + share of the population. The absolute floor is the recorded alternative. + +## Measured + +Method: the Python receipt below, run from the repo root against the release +`bible_vocab.txt` (`v0.1.0-cam96-data`, SHA-256 +`8dc3a65dcd3af38a2f53308fb96ef5fb5b34336c14587f6971b75ac966c3e212`). The +in-code gate G8c recomputes the band numbers and asserts them. + +| quantity | value | +|---|---| +| `word_forms.csv` rows with an empty `wordFreq` | 0 | +| in-vocab words with ≥1 reading | 3,556 | +| … whose readings fold to ≥2 parser states | 635 | +| … of those, lemma-table keys (register never read) | 494 | +| **band population** | **141** | +| top reading shares its state with another reading | 0 of 141 | +| rank indices (`rank_per_10000`, n = 141) | 35, 105 | +| cuts (lo, hi) | 72, 97 | +| share range | 50 … 99 | +| contested / leaning / decisive | 34 / 71 / 36 | +| absolute alternative "< 60 contested" | 20 | + +Because no top reading shares its state with another reading, reading share +and state share are equal for all 141 words: F8 changes nothing here. + +Other vocabularies (same receipt, different word set): +- The COCA 5k lemma vocab: population 0. Every word is a lemma-table key. +- All COCA surfaces: population 388, cuts (66, 94), bands 96 / 189 / 103; + "< 60" gives 62. + +```python +# receipt: python3 receipt.py (from the lance-graph root) +import csv,collections,sys +vs=set(l.strip() for l in open(sys.argv[1]) if l.strip()) +lem={} +for r in list(csv.reader(open("crates/deepnsm/word_frequency/lemmas_5k.csv")))[1:]: + lem.setdefault(r[1].lower(),r[2]) +fold=lambda p:{'n':'N','p':'N','v':'V','j':'J','a':'D','d':'D'}.get(p,'O') +rd=collections.defaultdict(list) +for r in list(csv.reader(open("crates/deepnsm/word_frequency/word_forms.csv")))[1:]: + w=r[5].lower() + if w in vs: rd[w].append((fold(r[2]),int(r[4]))) +pop=[w for w,rs in rd.items() if len({p for p,_ in rs})>=2 and w not in lem] +sh=sorted(max(c for _,c in rd[w])*100//sum(c for _,c in rd[w]) for w in pop) +n=len(sh); rk=lambda p:min(p*n//10000,n-1) +lo,hi=sh[rk(2500)],sh[rk(7500)] +print(n,lo,hi,sum(s=hi for s in sh)) +``` + +## Design + +In `bible_wave.rs` only: + +1. `enum CoverageBand { Decisive, Leaning, Contested }`. It is a separate + implementation from the planner's `NestedBands` quantile buckets + (`lance-graph-planner/src/nested_bands.rs`), which are cited as prior art + but live in a crate this one does not depend on. Doc comments say + "reading share" and never "confidence", "σ", "CI" or "reliability": + shares of rare and common words weigh equally. +2. `struct BandCuts { lo: u8, hi: u8, population: usize }`, computed once in + `Tagger::load` by `calibrate(&lemmas, &evidence, &vocab)` (the vocabulary gives each id's word for the lemma check and its length). + - Population: every id in `0..vocab.len()` that (a) is not a key of the + tagger's own `lemmas` map (as built, first row wins), (b) has known + coverage, and (c) has readings that fold to ≥2 parser states. + - Key: `coverage(id)[0]`. + - Rank rule: ndarray `rank_per_10000`, index `p·n / 10000` (integer + division) clamped to `n − 1`, with p = 2500 for `lo` and 7500 for `hi`. + Hand-rolled in a few lines, because deepnsm-v2 builds with no ndarray + dependency (`Cargo.toml:17-20`). This rule is not nearest-rank: the two + differ when p·n/10000 is an exact integer (n = 4 gives index 1 here, 0 + under nearest-rank). They agree at n = 141. + - Empty population: `None` (no cuts). The function must not panic for + n = 0 or n = 1; the unit-test fixtures build 1-5 word vocabularies. + - The calibrator reads `coverage` directly and never the tagger's + archaic/`Other` fallback. +3. Assignment: share < lo is Contested; share ≥ hi is Decisive; otherwise + Leaning. Words outside the population get `None`. The three reasons for + `None` (lemma-table, single state, unknown coverage) are merged + deliberately: a reader asks only "is there a band", and the tests keep the + reasons apart. +4. Storage: `bands: Vec>` indexed by `WordId` (at most + 65,536 entries, since `WordId` is `u16`). +5. No tag changes. `Tagger::pos` is untouched. `main` prints one `BANDS` line + (population, cuts, share range, counts per band), and G8c asserts it. +6. Integer only: no floating point in calibration. + +## Non-goals + +- Any change to `lexical.rs` (F1). +- Changing any tag, including an early exit on Decisive: role resolution + (D-LXC-2) is the reader that acts on bands. +- A word-frequency axis: blocked by D-LXC-5 (`bible_vocab.txt` has no counts). +- A streaming rolling floor: the evidence is loaded once. +- Landing bands in a V3 facet header (parent plan F10). The cuts would have + to travel with the bands, since a band means different things per + vocabulary. +- Configurable quantiles or more than three bands. +- The dormant lowercasing duplicate-key case in `lowercase_word_column`. + +## Pre-registered gates + +Numbering continues the parent plan's G1-G7. + +- G8a `cargo test` (deepnsm-v2) green; clippy `--all-targets -D warnings` and + `fmt --check` clean. +- G8b KJV run unchanged from D-LXC-1: 70,396 triples, 1,237 subjects, G6 = 25. +- G8c In `main`: population 141, cuts (72, 97), bands 34 / 71 / 36. This is a + regression pin for the release KJV vocabulary only. +- Unit tests. Each must be disable-verified red before landing. + - T1 The cuts are computed, not constant: a population shifted upward + raises both cuts. + - T2 On a mixed population, both Contested and Decisive are non-empty. + - T3 Three fixtures each get no band: a lemma-table word, a word whose + readings fold to one state, and a word with unknown coverage. Each has its + own disable run. + - T4 An empty population gives no cuts; a one-word population does not + panic. + - T5 Boundaries on explicit `BandCuts`: share == lo is Leaning, and + share == hi is Decisive. + - T6 Can stay silent: identical shares give zero Contested and all Decisive. + - T7 The key is reading share: a word with a 40% top reading and a larger + state sum is banded on 40. + - T8 The rank rule is `rank_per_10000`, not nearest-rank: at n = 4 the lo + index is 1. + +## Commit contents (one commit, in this order) + +1. `crates/deepnsm-v2/examples/bible_wave.rs`: code and tests. +2. This plan (it carries D-LXC-11 for the `added-plans-have-dids` check). +3. `.claude/board/entries/2026-09-30-deepnsm-v2-coverage-bands.md`, then + `python3 .claude/tools/entries_index.py --write`. +4. `SUPERSESSION-INDEX.md` regenerated last. +5. `AGENT_LOG.md` entry naming the council run. + +STATUS_BOARD, INTEGRATION_PLANS and PR_ARC_INVENTORY are deny-listed for the +agent. The entry carries a ready-to-paste STATUS_BOARD row for the operator: + +`| D-LXC-11 | coverage bands (population quartiles) in bible_wave | In PR | deepnsm-v2-coverage-bands-v1 |` + +## Open for the operator + +- Relative (this plan) vs absolute ("< 60") bands. Relative follows the stated + intent; absolute marks 20 instead of 34 words contested on the KJV + vocabulary, and 62 instead of 96 on all COCA surfaces. + +## Change ledger + +**v1 → v2 (5 savants: prior art, iron rules, code truth, cascade, different +views; 0 VIOLATES).** +- Rank rule taken from ndarray `rank_per_10000`; `NestedBands` cited. +- The proposed new id D-LXC-12 was dropped, because it duplicated D-LXC-5. +- `CoverageBand` naming. +- Relative bands kept, with T6 and a printed share range; the absolute + alternative is recorded with numbers. +- Reading-share key made explicit (F8) and measured. +- T3 split into three conditions. +- Wording kept free of statistics. +- Population defined by the tagger's own lemma map; calibrator uses no + fallback. +- `vocab.len()` passed in; n = 0/1 safe. +- G8 numbering continues the parent plan. +- Deny-list-compliant board path. +- "Rolling floor" reworded. + +**v2 → v3 (3 reviewers: overclaim, dilution/collapse, firewall; 0 BLOCK; +3 P1, 9 P2).** +- One rank rule named everywhere (it had been described both as nearest-rank + and as `rank_per_10000`; they differ). +- Receipt, method and asset digest added for every number. +- The "0 differ" result explained: no top reading shares its state. +- `CoverageBand` used consistently. +- Gates renamed G8a-c; G8c labelled a KJV-only pin. +- The tests are required to be disable-verified before landing; they are not + described as already verified. +- T5 uses explicit cuts. T6 asserts all Decisive. T4 gains n = 1. T8 added. +- Merged `None` reasons stated as a decision. +- The early exit kept visible as a non-goal. +- The undefined "variant A" removed. +- The dependency claim cited (`Cargo.toml:17-20`). +- Commit contents listed; the STATUS_BOARD row handed to the operator. diff --git a/crates/deepnsm-v2/examples/bible_wave.rs b/crates/deepnsm-v2/examples/bible_wave.rs index 96046129a..a2801eb5a 100644 --- a/crates/deepnsm-v2/examples/bible_wave.rs +++ b/crates/deepnsm-v2/examples/bible_wave.rs @@ -168,6 +168,43 @@ fn main() { ); println!("G6 PASS register pick moves {moved} in-vocabulary tags from first-wins (pinned 25)"); + // G8c (D-LXC-11) — coverage bands. The cuts are quartiles of the band + // population, calibrated at load. Reported only: no tag reads a band. + // The pinned numbers hold for the released `bible_vocab.txt` only. + let cuts = tagger.cuts.expect("KILL G8c: empty band population"); + let count = |b: CoverageBand| tagger.bands.iter().filter(|x| **x == Some(b)).count(); + let (contested, leaning, decisive) = ( + count(CoverageBand::Contested), + count(CoverageBand::Leaning), + count(CoverageBand::Decisive), + ); + let banded: Vec = (0..nsm.vocab.len()) + .filter_map(|i| WordId::try_from(i).ok()) + .filter_map(|id| band_share(&tagger.lemmas, &tagger.evidence, &nsm.vocab, id)) + .collect(); + let (min, max) = ( + banded.iter().min().copied().unwrap_or(0), + banded.iter().max().copied().unwrap_or(0), + ); + println!( + "BANDS population {}, cuts ({}, {}), shares {min}..{max}: contested {contested}, \ + leaning {leaning}, decisive {decisive}", + cuts.population, cuts.lo, cuts.hi + ); + assert_eq!( + ( + cuts.population, + cuts.lo, + cuts.hi, + contested, + leaning, + decisive + ), + (141, 72, 97, 34, 71, 36), + "KILL G8c: coverage bands moved from the pinned KJV-vocabulary layout" + ); + println!("G8c PASS coverage bands match the pinned KJV-vocabulary layout"); + // ── stream: verse index = version; FSM → SPO ── let mut stream = TemporalStream::new(); let mut all: Vec<(u64, Spo)> = Vec::new(); @@ -1061,6 +1098,109 @@ fn dominant_pos(evidence: &LexicalEvidence, id: WordId) -> Option { Some(coca_pos(&r.pos.as_char().to_string())) } +/// Where a word's dominant reading share sits in the population (D-LXC-11). +/// +/// The share is the dominant READING's cumulative coverage, `coverage(id)[0]`, +/// never a summed parser-state share. The cut points are quartiles of the +/// population, so each band holds a known part of it; a band is therefore +/// relative to the loaded vocabulary. Shares of rare and common words weigh +/// the same, so a band says where a word sits, not how much to trust it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum CoverageBand { + /// Share at or above the upper cut. + Decisive, + /// Share between the cuts. + Leaning, + /// Share below the lower cut. + Contested, +} + +/// The two cut points, calibrated once at load, and the population they came +/// from. A band means nothing without its cuts, so they travel together. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct BandCuts { + lo: u8, + hi: u8, + population: usize, +} + +/// ndarray `hpc::rolling_floor::rank_per_10000`: index `p·len/10000` +/// (integer division) clamped to the last index, `0` for an empty sample. +/// Restated here because this crate has no ndarray dependency. It is not +/// nearest-rank: the two differ when `p·len/10000` is a whole number. +fn rank_per_10000(len: usize, per_10000: usize) -> usize { + if len == 0 { + return 0; + } + (per_10000 * len / 10_000).min(len - 1) +} + +/// The quartile cuts of an ascending list of shares; `None` if it is empty. +fn cuts_from_sorted(shares: &[u8]) -> Option { + if shares.is_empty() { + return None; + } + Some(BandCuts { + lo: shares[rank_per_10000(shares.len(), 2500)], + hi: shares[rank_per_10000(shares.len(), 7500)], + population: shares.len(), + }) +} + +/// The band of one share under given cuts. +fn band_of(share: u8, cuts: BandCuts) -> CoverageBand { + if share < cuts.lo { + CoverageBand::Contested + } else if share >= cuts.hi { + CoverageBand::Decisive + } else { + CoverageBand::Leaning + } +} + +/// The share that places word `id` in the band population, or `None` if the +/// word is outside it: a lemma-table key (the register is never read for it), +/// a word with unknown coverage, or a word whose readings fold to one parser +/// state (nothing to decide between). Reads the register only, never the +/// tagger's archaic/`Other` fallback. +fn band_share( + lemmas: &HashMap, + evidence: &LexicalEvidence, + vocab: &PaletteVocab, + id: WordId, +) -> Option { + if lemmas.contains_key(vocab.word(id)?) { + return None; + } + let share = evidence.coverage(id).first().copied().flatten()?; + let mut states = evidence + .readings(id) + .iter() + .map(|r| coca_pos(&r.pos.as_char().to_string())); + let first = states.next()?; + states.any(|s| s != first).then_some(share) +} + +/// Calibrate the cuts over every vocabulary word and band each one. +fn calibrate( + lemmas: &HashMap, + evidence: &LexicalEvidence, + vocab: &PaletteVocab, +) -> (Option, Vec>) { + let shares: Vec> = (0..vocab.len()) + .map(|i| { + WordId::try_from(i) + .ok() + .and_then(|id| band_share(lemmas, evidence, vocab, id)) + }) + .collect(); + let mut sorted: Vec = shares.iter().flatten().copied().collect(); + sorted.sort_unstable(); + let cuts = cuts_from_sorted(&sorted); + let bands = shares.iter().map(|s| Some(band_of((*s)?, cuts?))).collect(); + (cuts, bands) +} + /// Lowercase the `word` column of `word_forms.csv`, leaving the header and the /// other fields as they are. `load_word_forms_csv` matches surfaces exactly and /// the corpus tokens are lowercased; three COCA surfaces (`True`, `False`, @@ -1092,6 +1232,10 @@ struct Tagger { lemmas: HashMap, evidence: LexicalEvidence, report: WordFormsReport, + /// D-LXC-11 cuts, `None` when the band population is empty. + cuts: Option, + /// D-LXC-11 band per `WordId`. Reported only; `pos` does not read it. + bands: Vec>, } impl Tagger { @@ -1111,10 +1255,13 @@ impl Tagger { .or_insert_with(|| coca_pos(pos)); } let (evidence, report) = load_word_forms_csv(&lowercase_word_column(forms_csv), vocab)?; + let (cuts, bands) = calibrate(&lemmas, &evidence, vocab); Ok(Self { lemmas, evidence, report, + cuts, + bands, }) } @@ -1296,4 +1443,140 @@ mod tests { let (_, raw) = load_word_forms_csv(&f, &v).unwrap(); assert_eq!(raw.unrouted, 1); } + + // ── D-LXC-11 coverage bands ── + + /// One two-state word per share: `n` carries `share` of 100, `v` the rest. + fn two_state_rows(words: &[(&str, u64)]) -> String { + let mut rows = String::new(); + for (i, (w, share)) in words.iter().enumerate() { + let k = 2 * i + 1; + rows.push_str(&format!("{k},{w},n,9,{share},{w}\n")); + rows.push_str(&format!("{},{w},v,9,{},{w}\n", k + 1, 100 - share)); + } + forms(&rows) + } + + fn load_words(lemmas_csv: &str, forms_csv: &str, words: &[&str]) -> (PaletteVocab, Tagger) { + let v = vocab(words); + let t = Tagger::load(lemmas_csv, forms_csv, &v).unwrap(); + (v, t) + } + + fn band(t: &Tagger, v: &PaletteVocab, w: &str) -> Option { + t.bands[usize::from(v.id(w).unwrap())] + } + + const SPREAD: [(&str, u64); 8] = [ + ("wa", 50), + ("wb", 60), + ("wc", 70), + ("wd", 80), + ("we", 90), + ("wf", 95), + ("wg", 98), + ("wh", 99), + ]; + + // T1 the cuts come from the population, not from constants + #[test] + fn cuts_move_with_the_population() { + let low = [50u8, 55, 60, 65, 70, 75, 80, 85]; + let high = low.map(|s| s + 10); + let (a, b) = ( + cuts_from_sorted(&low).unwrap(), + cuts_from_sorted(&high).unwrap(), + ); + assert!(b.lo > a.lo && b.hi > a.hi, "{a:?} -> {b:?}"); + } + + // T2 a mixed population has both Contested and Decisive words + #[test] + fn a_mixed_population_has_both_ends() { + let words: Vec<&str> = SPREAD.iter().map(|(w, _)| *w).collect(); + let (v, t) = load_words(NO_LEMMAS, &two_state_rows(&SPREAD), &words); + assert_eq!(t.cuts.unwrap().population, 8); + assert_eq!(band(&t, &v, "wa"), Some(CoverageBand::Contested)); + assert_eq!(band(&t, &v, "wh"), Some(CoverageBand::Decisive)); + } + + // T3 each exclusion keeps its word out of the population + #[test] + fn excluded_words_get_no_band() { + let mut rows = two_state_rows(&SPREAD); + rows.push_str("20,lx,n,9,60,lx\n21,lx,v,9,40,lx\n"); // lemma-table key + rows.push_str("22,one,n,9,60,one\n23,one,p,9,40,one\n"); // n+p: one state + rows.push_str("24,unk,n,9,60,unk\n25,unk,v,9,,unk\n"); // unknown count + let mut words: Vec<&str> = SPREAD.iter().map(|(w, _)| *w).collect(); + words.extend(["lx", "one", "unk"]); + let (v, t) = load_words("rank,lemma,PoS\n1,lx,v\n", &rows, &words); + assert_eq!(t.cuts.unwrap().population, 8, "only the SPREAD words count"); + for w in ["lx", "one", "unk"] { + assert_eq!(band(&t, &v, w), None, "{w}"); + } + // Anti-vacuity: each excluded word has readings that were loaded. + for w in ["lx", "one", "unk"] { + assert_eq!(t.evidence.readings(v.id(w).unwrap()).len(), 2, "{w}"); + } + } + + // T4 no population gives no cuts; one word does not panic + #[test] + fn empty_and_single_populations() { + let (_, t) = load_words(NO_LEMMAS, &forms("1,x,n,9,5,w\n"), &["w"]); + assert_eq!(t.cuts, None); + assert_eq!(t.bands, vec![None]); + let (v, t) = load_words(NO_LEMMAS, &two_state_rows(&[("w", 70)]), &["w"]); + assert_eq!( + t.cuts, + Some(BandCuts { + lo: 70, + hi: 70, + population: 1 + }) + ); + assert_eq!(band(&t, &v, "w"), Some(CoverageBand::Decisive)); + } + + // T5 the boundaries, on explicit cuts + #[test] + fn band_boundaries() { + let c = BandCuts { + lo: 50, + hi: 80, + population: 0, + }; + assert_eq!(band_of(49, c), CoverageBand::Contested); + assert_eq!(band_of(50, c), CoverageBand::Leaning); + assert_eq!(band_of(79, c), CoverageBand::Leaning); + assert_eq!(band_of(80, c), CoverageBand::Decisive); + } + + // T6 can stay silent: identical shares mark nothing Contested + #[test] + fn identical_shares_are_all_decisive() { + let c = cuts_from_sorted(&[70; 8]).unwrap(); + assert_eq!((c.lo, c.hi), (70, 70)); + assert_eq!(band_of(70, c), CoverageBand::Decisive); + } + + // T7 the key is the dominant READING's share, never a state sum + #[test] + fn the_key_is_reading_share() { + let f = forms("1,x,n,9,40,w\n2,y,p,9,35,w\n3,z,v,9,25,w\n"); + let (v, t) = load_words(NO_LEMMAS, &f, &["w"]); + let id = v.id("w").unwrap(); + // n 40 + p 35 would be Noun 75 if summed. + assert_eq!(band_share(&t.lemmas, &t.evidence, &v, id), Some(40)); + } + + // T8 the rank rule is ndarray's rank_per_10000, not nearest-rank + #[test] + fn rank_rule_is_rank_per_10000() { + assert_eq!(rank_per_10000(4, 2500), 1); // nearest-rank: ceil(1) - 1 = 0 + assert_eq!(rank_per_10000(141, 2500), 35); + assert_eq!(rank_per_10000(141, 7500), 105); + assert_eq!(rank_per_10000(0, 2500), 0); + assert_eq!(rank_per_10000(1, 7500), 0); + } } From 9b830eb16a4242f28556824c484a75015ec2b8c0 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 30 Sep 2026 02:55:38 +0000 Subject: [PATCH 07/11] plan: D-LXC-11 results and disable table Every new test T1-T8 turned red under its own disable, run after the implementation commit. Claude-Session: https://claude.ai/code/session_01AUbvJZf4pD6GzkZZ6LtqSE --- .claude/plans/deepnsm-v2-coverage-bands-v1.md | 23 ++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/.claude/plans/deepnsm-v2-coverage-bands-v1.md b/.claude/plans/deepnsm-v2-coverage-bands-v1.md index 07342b652..1172ecaf1 100644 --- a/.claude/plans/deepnsm-v2-coverage-bands-v1.md +++ b/.claude/plans/deepnsm-v2-coverage-bands-v1.md @@ -27,7 +27,7 @@ floor; the streaming form exists as ndarray `hpc::rolling_floor::RollingFloor`. - [x] Spec v1, 5 research savants, draft v2, 3 reviewers, v3 (this file) - [x] `CoverageBand`, `BandCuts`, `calibrate` in `bible_wave.rs` -- [ ] Tests T1-T8, each disable-verified red before landing +- [x] Tests T1-T8, each disable-verified red before landing - [x] Gates G8a-G8c - [x] Board entry + indexes in the same commit @@ -191,6 +191,27 @@ agent. The entry carries a ready-to-paste STATUS_BOARD row for the operator: `| D-LXC-11 | coverage bands (population quartiles) in bible_wave | In PR | deepnsm-v2-coverage-bands-v1 |` +## Results (2026-09-30) + +- G8a: 126 lib + 17 example tests pass; clippy `-D warnings` and fmt clean. +- G8b: KJV unchanged — 70,396 triples, 1,237 subjects, G6 = 25. +- G8c: population 141, cuts (72, 97), shares 50..99, bands 34/71/36; equal + to the receipt. +- Disable runs, each after the commit, each restored with `git checkout`: + +| disable | red | +|---|---| +| constant cuts (72, 97) | T1, T4, T6 | +| no empty-population guard | T4 and 5 older tests (index panic on n = 0) | +| no lemma-table exclusion | T3 | +| unknown coverage read as 100 | T3 | +| no two-state requirement | T3, T4 | +| `share <= lo` is Contested | T5, T4, T6 | +| `share > hi` is Decisive | T5, T4, T6 | +| key = summed state share | T7 | +| nearest-rank instead of `rank_per_10000` | T8 | +| every word Leaning | T2, T4, T5, T6 | + ## Open for the operator - Relative (this plan) vs absolute ("< 60") bands. Relative follows the stated From 7bf1873920cd2290e20b2bb2c6a7984a6cd4ff75 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 30 Sep 2026 21:21:30 +0000 Subject: [PATCH 08/11] deepnsm-v2: frequency is evidence, not a lexical decision Remove dominant_pos and every path where readings()[0] became the tag. Tagger::pos reads load_pos_legacy_first_wins (main's load_pos, renamed so the compatibility boundary is visible), then archaic_pos, then Other. It reads no count. Its source-row-order dependence is inherited debt pending D-LXC-2/D-LXC-3, not an authorized resolver. Kept: count-ordered storage, cumulative percentile coverage, the D-LXC-11 band population, rank rule and cuts (report vocabulary only), the storage tests and the reproducibility receipts. G6 (25 moved tags) is replaced by a paired non-interference invariant: counts_change_evidence_never_the_readings_or_the_tag. Count changes may move order and coverage, never the reading set or the tag. A disable run that re-derives the tag from readings()[0] turns exactly that test red. KJV: 70,393 triples, 1,227 subjects, 1,941 predicates, identical to main. Docs: position 0 is the most frequent observed reading, not a preference. Plans get an appended correction; board entries get appended corrections. Claude-Session: https://claude.ai/code/session_01AUbvJZf4pD6GzkZZ6LtqSE --- ...9-29-deepnsm-v2-counted-pick-tag-deltas.md | 12 + .../2026-09-30-deepnsm-v2-coverage-bands.md | 7 + .claude/plans/deepnsm-v2-coverage-bands-v1.md | 29 ++- ...deepnsm-v2-lexical-evidence-consumer-v1.md | 75 +++++- crates/deepnsm-v2/examples/bible_wave.rs | 228 ++++++++---------- crates/deepnsm-v2/src/lexical.rs | 26 +- 6 files changed, 235 insertions(+), 142 deletions(-) diff --git a/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md b/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md index db87f2807..484138ff6 100644 --- a/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md +++ b/.claude/board/entries/2026-09-29-deepnsm-v2-counted-pick-tag-deltas.md @@ -48,3 +48,15 @@ percentile coverage, and the tagger reads position 0. The KJV run is identical to "after (B)" above (25 moved, 70,396 triples); the summing moved nothing. Plan: `.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md`. + +**Correction (2026-09-30, operator ruling on #1304).** Frequency is evidence, +not a lexical decision. The 25 moved tags and the 70,396 triples above record +frequency changing the tag: unintended semantic interference, not a +correction. The counted pick and `dominant_pos` are removed. The tag is +`main`'s tagging again, named `load_pos_legacy_first_wins`; its dependence on +source-row order is inherited debt pending D-LXC-2/D-LXC-3, not an authorized +resolver. KJV after the repair: 70,393 triples, 1,227 subjects, 1,941 +predicates — identical to `main`. G6 is now a paired invariant pinned by +`counts_change_evidence_never_the_readings_or_the_tag`: count changes may move +the order and the coverage, never the reading set or the tag. The table above +is kept as the historical measurement. diff --git a/.claude/board/entries/2026-09-30-deepnsm-v2-coverage-bands.md b/.claude/board/entries/2026-09-30-deepnsm-v2-coverage-bands.md index e994f59a3..3ed982f03 100644 --- a/.claude/board/entries/2026-09-30-deepnsm-v2-coverage-bands.md +++ b/.claude/board/entries/2026-09-30-deepnsm-v2-coverage-bands.md @@ -31,3 +31,10 @@ Ready-to-paste STATUS_BOARD row (the file is deny-listed for the agent): `| D-LXC-11 | coverage bands (population quartiles) in bible_wave | In PR | deepnsm-v2-coverage-bands-v1 |` Plan: `.claude/plans/deepnsm-v2-coverage-bands-v1.md`. + +**Correction (2026-09-30).** The bands are a measurement, not a decision: the +labels are report vocabulary in `bible_wave`, nothing reads a band to select +or drop a reading, and no downstream consumer exists. The numbers (141, +(72, 97), 34 / 71 / 36) stand. "70,396 triples, G6 = 25" above is stale: after +the #1304 repair the KJV run is 70,393 triples, identical to `main`, and G6 is +the non-interference invariant (see the 2026-09-29 entry's correction). diff --git a/.claude/plans/deepnsm-v2-coverage-bands-v1.md b/.claude/plans/deepnsm-v2-coverage-bands-v1.md index 1172ecaf1..4becfc46a 100644 --- a/.claude/plans/deepnsm-v2-coverage-bands-v1.md +++ b/.claude/plans/deepnsm-v2-coverage-bands-v1.md @@ -1,6 +1,11 @@ # DeepNSM-v2 coverage bands at population-calibrated thresholds (D-LXC-11) **Status:** RATIFIED v3 (5+3 council, 2026-09-30). Implementation in this PR. +**Reframed 2026-09-30 (operator):** the bands are a MEASUREMENT, not a +decision. Keep the number; be suspicious of the adjective. The labels +Decisive / Leaning / Contested are report vocabulary for `bible_wave` only — +nothing reads a band to select, rank or eliminate a reading, and no +downstream consumer exists. See "Reframe" below. **Parent:** `.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md` (D-LXC-1, the frequency-ordered register this reads). @@ -17,7 +22,7 @@ percentile coverage with thresholds at prevalence boundaries: > Akin to using Gini palette when looking at poor countries. Or IQ buckets in > gaussian distribution." -This plan turns the dominant reading's share into a three-level +This plan turns the most frequent reading's share into a three-level `CoverageBand` whose cut points are quartiles of the measured population, so each band holds a known share of it. The cuts are calibrated once at load from the evidence present. That is a load-time snapshot, not a streaming rolling @@ -46,7 +51,7 @@ floor; the streaming form exists as ndarray `hpc::rolling_floor::RollingFloor`. - F6 No model identifiers or advertising; Read for source inspection; the board files denied in `.claude/settings.json` are not edited by the agent. - F7 Search is navigation, never evidence. -- F8 The band key is the dominant READING's share, `coverage(id)[0]`. The +- F8 The band key is the most frequent READING's share, `coverage(id)[0]`. The operator rejected summing at tag time ("why are you counting … it should simply be percentile coverage"), so no parser-state sum is formed. - F9 Bands are relative by design: the operator asked for cuts at prevalence @@ -141,8 +146,10 @@ In `bible_wave.rs` only: ## Non-goals - Any change to `lexical.rs` (F1). -- Changing any tag, including an early exit on Decisive: role resolution - (D-LXC-2) is the reader that acts on bands. +- Changing any tag, including an early exit on Decisive. ⊘ 2026-09-30: + "role resolution (D-LXC-2) is the reader that acts on bands" is struck. A + band describes how concentrated a word's evidence is; it is never a licence + to drop a reading. Only structural/contextual evidence eliminates one. - A word-frequency axis: blocked by D-LXC-5 (`bible_vocab.txt` has no counts). - A streaming rolling floor: the evidence is loaded once. - Landing bands in a V3 facet header (parent plan F10). The cuts would have @@ -158,6 +165,9 @@ Numbering continues the parent plan's G1-G7. - G8a `cargo test` (deepnsm-v2) green; clippy `--all-targets -D warnings` and `fmt --check` clean. - G8b KJV run unchanged from D-LXC-1: 70,396 triples, 1,237 subjects, G6 = 25. + ⊘ 2026-09-30: stale — those numbers were frequency changing tags. After the + repair: 70,393 triples, 1,227 subjects (identical to `main`), and G6 is the + non-interference invariant (parent plan, Correction). - G8c In `main`: population 141, cuts (72, 97), bands 34 / 71 / 36. This is a regression pin for the release KJV vocabulary only. - Unit tests. Each must be disable-verified red before landing. @@ -177,6 +187,17 @@ Numbering continues the parent plan's G1-G7. - T8 The rank rule is `rank_per_10000`, not nearest-rank: at n = 4 the lo index is 1. +## Reframe (2026-09-30) + +- Kept verbatim: the population definition, `rank_per_10000`, the cuts, + G8c (141, (72, 97), 34 / 71 / 36) and T1-T8. +- The labels stay as report vocabulary. No rename, no new type: renaming + would be an abstraction for its own sake, and nothing consumes them. +- `CoverageBand`'s doc says so: the statistic is `coverage(id)[0]`, and a + band selects nothing. +- A future consumer that wants to act on a band must first show a + structural reason; the band alone is never one. + ## Commit contents (one commit, in this order) 1. `crates/deepnsm-v2/examples/bible_wave.rs`: code and tests. diff --git a/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md b/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md index 58a676c71..8053cb649 100644 --- a/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md +++ b/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md @@ -1,6 +1,8 @@ # deepnsm-v2-lexical-evidence-consumer-v1 -**Status:** PROPOSAL. No code authorized. Written against `main` `5282dfa3` +**Status:** CORRECTED 2026-09-30 — see "Correction: frequency is evidence, +not a lexical decision" below; it supersedes the counted pick, G3(b)-(f) and +G6. Originally: PROPOSAL. No code authorized. Written against `main` `5282dfa3` (the #1299 merge). Rewritten 2026-09-29 after every claim was re-read in source with the Read tool; the earlier council version relied on shell searches and on files it never opened (change ledger at the end). @@ -64,6 +66,9 @@ Read in full before this rewrite: false. The file is ordered by lemma rank, not by `wordFreq`, and 259 surfaces have a first row that is not their most frequent reading (measured below). D-LXC-1 fixes exactly that premise and keeps the lemma-first rule. + ⊘ 2026-09-30: struck. The premise is false, but replacing it with "the most + frequent row wins" is the same kind of resolver. D-LXC-1 now measures the + reading order and does not tag from it (see the Correction). - **v1 prior art:** `deepnsm/src/vocabulary.rs:131-135,158-184` keeps the first occurrence for lemmas and adds forms only when absent (first-wins, no counts). `deepnsm/examples/homograph_collapse.rs:3-15,84-92` selects a @@ -101,7 +106,8 @@ Read in full before this rewrite: - **Surface-keyed, counted:** `word_forms.csv` through `LexicalEvidence`, every reading of a surface with its `wordFreq`. -Neither is a role-based reading. The counted pick is an interim leg. The +Neither is a role-based reading. ⊘ 2026-09-30: "The counted pick is an +interim leg" is struck — frequency may not pick a reading, interim or not. The role-based route (`E-SURFACE-FORM-COLLAPSE-1`) needs several readings per token in the FSM, which is D-LXC-2. @@ -223,6 +229,8 @@ runs. The largest are bases 16, locks 15, promises 13 and flies 10. - **`WordFormsReport`:** 11,460 rows, 4,284 readings stored, 4 empty surface, 7,172 not in the KJV vocabulary. +- ⊘ 2026-09-30: the reading below is struck. The 25 moved tags and 3 added + triples were frequency changing semantics, which the Correction forbids. - **Reading.** B is a small correction. It moves 25 words' tags, adds 3 triples, and leaves the long-range shares unchanged. Whether the moved tags are more correct is still not measured. A moves 5.6× more tags and adds 695 @@ -233,6 +241,9 @@ runs. ### Amendment: frequency is the register (2026-09-29, operator-directed) +⊘ 2026-09-30: the storage order and coverage below stand; "frequency is the +register's preference" and `dominant_pos` are struck (see the Correction). + The counted pick summed counts at tag time. Frequency is a dimension, so it belongs in the storage order, expressed as percentile coverage: - `LexicalEvidenceBuilder::finish` stores each word's readings most frequent @@ -248,8 +259,65 @@ belongs in the storage order, expressed as percentile coverage: - This supersedes the "no library change" non-goal (F1): the change is storage order plus a derived integer, not interpretation. +## Correction: frequency is evidence, not a lexical decision (2026-09-30) + +Operator ruling on #1304. Frequency calibrates ambiguity; it never chooses, +ranks for selection, or eliminates a reading. Neither count magnitude nor +source-row order may choose meaning or PoS. + +**Kept (measurement):** the count-ordered storage, unknown-last and stable +ties; cumulative percentile coverage with `u128` arithmetic; the D-LXC-11 +band population, rank rule and cuts; lowercasing; the `test = true` home. + +**Removed (selection):** `dominant_pos` and every path where +`readings()[0]` becomes the tag. `Tagger::pos` now reads +`load_pos_legacy_first_wins` → `archaic_pos` → `Pos::Other` — `main`'s +`load_pos` verbatim, renamed so the boundary is visible. It reads no count. + +**Inherited debt, not precedent:** the legacy map still depends on +source-row order (the first lemma row, the first form row). That is not an +authorized resolver; it stays until D-LXC-2 (several readings into the +parser) and D-LXC-3 (the lemma-table migration). F9 is unchanged in #1304 +and is documented as this same inherited exception. + +**G6 replaced.** The historical G6 = 25 (and 70,396 triples) measured +frequency changing tags — unintended semantic interference, not a +correction. The new G6 is a paired non-interference invariant, pinned by +`counts_change_evidence_never_the_readings_or_the_tag`: +- count changes MUST NOT change the reading set or the tag; +- count changes MAY change the ordering, coverage and statistics. +A disable run that re-derives the tag from `readings()[0]` turns exactly +that test red. KJV after the repair: 70,393 triples, 1,227 subjects, +1,941 predicates — identical to `main`. + +**Gates now:** G3(a), (g), (h), (i) stand. G3(b)-(f) tested the counted +pick and are retired with it; their tests were deleted, and (f) became +`legacy_tagging_falls_through_to_archaic_then_other`. G6 is the invariant +above. G5 stands with the numbers above. + +**Multiplicity.** #1304 stops at exact readings plus calibrated evidence. +No parser fold, `PosSet`, reading mask, Det/Adj licensing, or ambiguity +marking in the `Spo` high bits — the reading-set ABI boundary is specified +first. + +### Sizing evidence for D-LXC-2 / D-LXC-3 (not a design) + +Python simulation of a reading-set fold over the parser's three control +states, run on the KJV verse stream with the same vocabulary; not committed. +- F9 kept: 3,363 tokens carry more than one folded reading; at most 4 live + parser configurations at once; 4,304 alternative triples. +- A Det/Adj licensing probe resolves 1,895 of those tokens (56%), giving + 71,400 triples. +- F9 lifted: 91,054 multi-reading tokens; peak 17 live configurations. +- At most 6 readings per surface, so a `u8` reading mask over a word's span + fits. +- Configurations are bounded per clause (polynomial), not one register + tuple per state. + ## Pre-registered gates +⊘ 2026-09-30: G3(b)-(f) and G6 below are superseded by the Correction above. + - G1 `cargo test --manifest-path crates/deepnsm-v2/Cargo.toml` green; a disable run under exactly that command turns the new tests red. - G2 clippy `--all-targets -D warnings` and `fmt --check` clean. @@ -313,7 +381,8 @@ belongs in the storage order, expressed as percentile coverage: 1. **D-LXC-1** · Lexical-evidence consumer (this plan). 2. **D-LXC-2** · The FSM takes several readings per token and resolves them by role (`E-SURFACE-FORM-COLLAPSE-1`). Needs an FSM change; separate PR. -3. **D-LXC-3** · Lemma-table order. B keeps F9 (25 in-vocabulary changes); +3. **D-LXC-3** · Lemma-table order (⊘ 2026-09-30: neither A nor B may pick + by count; the migration target is a reading set, not a new order). B keeps F9 (25 in-vocabulary changes); A drops it (141). Switching to A overrides F9 and is the operator's call, taken on the D-LXC-1 KJV numbers. 4. **D-LXC-4** · `academic_20k.csv` loader, plus the duplicate `coca_pos` in diff --git a/crates/deepnsm-v2/examples/bible_wave.rs b/crates/deepnsm-v2/examples/bible_wave.rs index a2801eb5a..b4eba4056 100644 --- a/crates/deepnsm-v2/examples/bible_wave.rs +++ b/crates/deepnsm-v2/examples/bible_wave.rs @@ -12,8 +12,10 @@ //! cargo run --example bible_wave -- /path/to/pg10.txt //! ``` //! -//! Pipeline: verses → PoS-tag (COCA lemma table, then the counted -//! `word_forms.csv` evidence, then a documented archaic fallback) → FSM → SPO +//! Pipeline: verses → PoS-tag (legacy single-`Pos` tagging: COCA lemma table, +//! then the first `word_forms.csv` row, then a documented archaic fallback; +//! the counted `word_forms.csv` evidence is loaded beside it for measurement +//! only and chooses no tag) → FSM → SPO //! stream (verse index = version) → `TemporalStream` + //! the TRAINED Cam96 codebook (`data/`, real Jina-v3 embeddings). //! @@ -143,33 +145,15 @@ fn main() { "LEXICON word_forms: {} rows, {} readings stored, {} empty surface, {} not in vocab", r.rows, r.stored, r.empty_surface, r.unrouted ); - // G6 (D-LXC-1) — how many in-vocabulary tags the register pick moves away - // from the old first-wins rule. Pinned against the released - // `bible_vocab.txt`: any other number means the tables or the rule changed. - let first_wins = load_pos_first_wins(&lemmas_csv, &forms_csv); - let mut moved = 0usize; - for id in 0..nsm.vocab.len() { - let id = WordId::try_from(id).expect("vocab fits u16"); - let Some(w) = nsm.vocab.word(id) else { - continue; - }; - let old = first_wins - .get(w) - .copied() - .or_else(|| archaic_pos(w)) - .unwrap_or(Pos::Other); - if tagger.pos(w, id) != old { - moved += 1; - } - } - assert_eq!( - moved, 25, - "KILL G6: the register pick moved {moved} in-vocabulary tags, pinned 25" - ); - println!("G6 PASS register pick moves {moved} in-vocabulary tags from first-wins (pinned 25)"); - - // G8c (D-LXC-11) — coverage bands. The cuts are quartiles of the band - // population, calibrated at load. Reported only: no tag reads a band. + // G6 (D-LXC-1) is the non-interference invariant, proven by the focused + // test `counts_change_evidence_never_the_readings_or_the_tag`: counts move + // the evidence order and coverage, never the reading set or the tag. The + // historical "25 moved tags" measured count leaking into the tag; see the + // board entry of 2026-09-29. + + // G8c (D-LXC-11) — coverage bands, a MEASUREMENT of reading concentration. + // The cuts are quartiles of the band population, calibrated at load. + // Reported only: no tag reads a band. // The pinned numbers hold for the released `bible_vocab.txt` only. let cuts = tagger.cuts.expect("KILL G8c: empty band population"); let count = |b: CoverageBand| tagger.bands.iter().filter(|x| **x == Some(b)).count(); @@ -214,7 +198,7 @@ fn main() { for tok in verse.split_whitespace() { let Some(w) = normalise(tok) else { continue }; let Some(id) = nsm.vocab.id(&w) else { continue }; - let pos = tagger.pos(&w, id); + let pos = tagger.pos(&w); tagged_buf.push(Tagged::new(id, pos)); } tagged_buf.push(Tagged::new(0, Pos::Stop)); // verse boundary flushes @@ -1081,30 +1065,16 @@ fn normalise(tok: &str) -> Option { (w.len() >= 2).then_some(w) } -/// The dominant reading of one word (D-LXC-1): position 0 of the -/// frequency-ordered register, folded with [`coca_pos`]. -/// -/// `LexicalEvidence` stores readings most frequent first, so nothing is summed -/// here. The register is read only when its coverage is known; with any -/// unknown count the dominant reading is unknown and this returns `None`, so -/// the caller's fallback decides. -/// -/// This replaces taking the FIRST `word_forms.csv` row: that file is ordered by -/// lemma rank, not by surface frequency, so for 259 surfaces the first row is -/// not the dominant reading (`changes`: verb row 13,624 first, noun 113,085). -fn dominant_pos(evidence: &LexicalEvidence, id: WordId) -> Option { - evidence.coverage(id).first().copied().flatten()?; - let r = evidence.readings(id).first()?; - Some(coca_pos(&r.pos.as_char().to_string())) -} - -/// Where a word's dominant reading share sits in the population (D-LXC-11). +/// Where a word's reading concentration sits in the population (D-LXC-11). /// -/// The share is the dominant READING's cumulative coverage, `coverage(id)[0]`, -/// never a summed parser-state share. The cut points are quartiles of the -/// population, so each band holds a known part of it; a band is therefore -/// relative to the loaded vocabulary. Shares of rare and common words weigh -/// the same, so a band says where a word sits, not how much to trust it. +/// A MEASUREMENT, not a decision. The statistic is the share of the most +/// frequent observed reading, `coverage(id)[0]`, never a summed parser-state +/// share. The cut points are quartiles of the population, so each band holds a +/// known part of it; a band is therefore relative to the loaded vocabulary. +/// Shares of rare and common words weigh the same, so a band says where a +/// word's concentration sits, not how much to trust it and not which reading +/// holds. The labels are report vocabulary for this example only: nothing +/// reads a band to select, rank or eliminate a reading. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum CoverageBand { /// Share at or above the upper cut. @@ -1223,13 +1193,19 @@ fn lowercase_word_column(forms_csv: &str) -> String { out } -/// The corpus tagger: lemma table → dominant `word_forms.csv` reading → -/// [`archaic_pos`] → [`Pos::Other`]. +/// The corpus tagger, plus the lexical evidence and its calibration. /// -/// Lemma-first is deliberate and pinned (commit `ec50f07b`): the forms layer -/// only fills silence, never overrules a word the lemma table knew. +/// The tag ([`Tagger::pos`]) is the LEGACY single-`Pos` tagging `main` had +/// before D-LXC-1, unchanged ([`load_pos_legacy_first_wins`]). The parser takes +/// one `Pos` per token, so a word with several readings cannot reach it +/// intact; this boundary is inherited debt kept for compatibility, not a +/// semantic resolution. The evidence and the bands are measurements beside it: +/// no count, order or band is read to produce a tag. struct Tagger { lemmas: HashMap, + /// `main`'s tagging, kept verbatim: lemma table, then first + /// `word_forms.csv` row. See [`load_pos_legacy_first_wins`]. + legacy: HashMap, evidence: LexicalEvidence, report: WordFormsReport, /// D-LXC-11 cuts, `None` when the band population is empty. @@ -1258,6 +1234,7 @@ impl Tagger { let (cuts, bands) = calibrate(&lemmas, &evidence, vocab); Ok(Self { lemmas, + legacy: load_pos_legacy_first_wins(lemmas_csv, forms_csv), evidence, report, cuts, @@ -1265,21 +1242,29 @@ impl Tagger { }) } - /// The tag for word `w` whose routing id is `id`. - fn pos(&self, w: &str, id: WordId) -> Pos { - if let Some(&p) = self.lemmas.get(w) { - return p; - } - dominant_pos(&self.evidence, id) + /// The LEGACY tag for word `w`: [`load_pos_legacy_first_wins`], then + /// [`archaic_pos`], then [`Pos::Other`] — exactly `main`'s tagging. + /// + /// Compatibility boundary, not semantic resolution. It reads no count and + /// no [`LexicalEvidence`], so frequency cannot change a tag. It does + /// depend on source-row order (the first lemma row, the first form row); + /// that is inherited debt, not an authorized resolver, pending a + /// multi-reading parser input (D-LXC-2) and the lemma-table migration + /// (D-LXC-3). + fn pos(&self, w: &str) -> Pos { + self.legacy + .get(w) + .copied() .or_else(|| archaic_pos(w)) .unwrap_or(Pos::Other) } } -/// The tagging `bible_wave` used before D-LXC-1: the lemma table, then the -/// FIRST `word_forms.csv` row per surface. Kept only so gate G6 can count how -/// many tags the counted pick moves. -fn load_pos_first_wins(lemmas_csv: &str, forms_csv: &str) -> HashMap { +/// LEGACY tagging, verbatim from `main`'s `load_pos`: the lemma table, then the +/// FIRST `word_forms.csv` row per surface, both first row wins. Row order is +/// not an authorized lexical decision; this is kept only because the parser +/// takes one `Pos` per token (see [`Tagger::pos`]). +fn load_pos_legacy_first_wins(lemmas_csv: &str, forms_csv: &str) -> HashMap { let mut m: HashMap = HashMap::new(); for line in lemmas_csv.lines().skip(1) { let f: Vec<&str> = line.split(',').collect(); @@ -1330,7 +1315,7 @@ mod tests { fn tag(forms_csv: &str, w: &str) -> Pos { let v = vocab(&[w]); let t = Tagger::load(NO_LEMMAS, forms_csv, &v).unwrap(); - t.pos(w, v.id(w).unwrap()) + t.pos(w) } // (a) a homograph keeps every reading @@ -1347,57 +1332,16 @@ mod tests { assert!(tags.contains(&'n') && tags.contains(&'v'), "{tags:?}"); } - // (b) the pick is position 0 of the register: folded tags are NOT summed - #[test] - fn the_dominant_reading_wins_without_summing() { - // n 60 + p 50 would be Noun 110 if summed; the register says v 100. - assert_eq!( - tag(&forms("1,x,n,9,60,w\n2,y,p,9,50,w\n3,z,v,9,100,w\n"), "w"), - Pos::Verb - ); - } - - // (c) an unknown count makes the register unreadable, so the fallback decides + // (f) the LEGACY fallback order, pinned as inherited debt, not endorsed: + // legacy map (a form row beats archaic, D-LXC-9), then archaic, then Other. #[test] - fn an_unknown_count_falls_through() { - assert_eq!(tag(&forms("1,x,n,9,500,w\n2,y,v,9,,w\n"), "w"), Pos::Other); - } - - // (d) `changes`: first row is the verb, the register says noun (89%) - #[test] - fn changes_is_a_noun_by_frequency_and_a_verb_by_first_row() { - let forms = committed("word_forms.csv"); - let v = vocab(&["changes"]); - let t = Tagger::load(NO_LEMMAS, &forms, &v).unwrap(); - let id = v.id("changes").unwrap(); - assert!( - !t.lemmas.contains_key("changes"), - "must not be a lemma-table key" - ); - assert_eq!(t.pos("changes", id), Pos::Noun); - assert_eq!(t.evidence.coverage(id).first().copied().flatten(), Some(89)); - // Anti-vacuity: the old first-wins rule tags it Verb. - assert_eq!( - load_pos_first_wins(NO_LEMMAS, &forms).get("changes"), - Some(&Pos::Verb) - ); - } - - // (e) equal counts keep file order - #[test] - fn ties_keep_file_order() { - assert_eq!(tag(&forms("1,x,v,9,10,w\n2,y,n,9,10,w\n"), "w"), Pos::Verb); - assert_eq!(tag(&forms("1,x,n,9,10,w\n2,y,v,9,10,w\n"), "w"), Pos::Noun); - } - - // (f) no known count falls through to archaic, then Other - #[test] - fn no_known_count_falls_through_to_archaic_then_other() { - let f = forms("1,hath,n,5,,hath\n2,zz,n,5,,zz\n"); - let v = vocab(&["hath", "zz"]); + fn legacy_tagging_falls_through_to_archaic_then_other() { + let f = forms("1,art,n,5,,art\n"); + let v = vocab(&["art", "hath", "zz"]); let t = Tagger::load(NO_LEMMAS, &f, &v).unwrap(); - assert_eq!(t.pos("hath", v.id("hath").unwrap()), Pos::Verb); - assert_eq!(t.pos("zz", v.id("zz").unwrap()), Pos::Other); + assert_eq!(t.pos("art"), Pos::Noun); + assert_eq!(t.pos("hath"), Pos::Verb); + assert_eq!(t.pos("zz"), Pos::Other); } // (g) stay-silent: one reading keeps its tag and covers 100% @@ -1418,17 +1362,17 @@ mod tests { ) .unwrap(); let id = v.id("work").unwrap(); - // The register alone says Noun ... - assert_eq!(dominant_pos(&t.evidence, id), Some(Pos::Noun)); - // ... and the lemma table still wins. - assert_eq!(t.pos("work", id), Pos::Verb); + // The most frequent observed reading is the noun ... + assert_eq!(t.evidence.readings(id)[0].pos.as_char(), 'n'); + // ... and the tag is still the lemma table's. + assert_eq!(t.pos("work"), Pos::Verb); // The same property on a conflicting row, with the row proven loaded. let conflicting = forms("1,x,n,9,4,create\n"); let v = vocab(&["create"]); let t = Tagger::load("rank,lemma,PoS\n1,create,v\n", &conflicting, &v).unwrap(); assert_eq!(t.evidence.reading_count(), 1); - assert_eq!(t.pos("create", v.id("create").unwrap()), Pos::Verb); + assert_eq!(t.pos("create"), Pos::Verb); } // (i) a capitalised COCA surface still routes @@ -1438,12 +1382,48 @@ mod tests { let v = vocab(&["true"]); let t = Tagger::load(NO_LEMMAS, &f, &v).unwrap(); assert_eq!(t.report.unrouted, 0); - assert_eq!(t.pos("true", v.id("true").unwrap()), Pos::Adj); + assert_eq!(t.pos("true"), Pos::Adj); // Without the lowercasing the same row is not routed. let (_, raw) = load_word_forms_csv(&f, &v).unwrap(); assert_eq!(raw.unrouted, 1); } + // Frequency is evidence, not a lexical decision. Changing only the counts, + // with the same rows in the same order, MAY change the evidence order and + // coverage; it MUST NOT change the reading set or the tag. + #[test] + fn counts_change_evidence_never_the_readings_or_the_tag() { + let v = vocab(&["w"]); + let id = v.id("w").unwrap(); + let load = |rows: &str| Tagger::load(NO_LEMMAS, &forms(rows), &v).unwrap(); + // `changes`' real counts (verb row first), then swapped, then even. + let real = load("1,x,v,9,13624,w\n2,y,n,9,113085,w\n"); + let swapped = load("1,x,v,9,113085,w\n2,y,n,9,13624,w\n"); + let even = load("1,x,v,9,50,w\n2,y,n,9,50,w\n"); + let set = |t: &Tagger| { + let mut s: Vec<_> = t + .evidence + .readings(id) + .iter() + .map(|r| (r.pos, r.lemma)) + .collect(); + s.sort(); + s + }; + // MAY change: the evidence order and the coverage. + assert_ne!( + real.evidence.readings(id)[0].pos, + swapped.evidence.readings(id)[0].pos + ); + assert_eq!(real.evidence.coverage(id), &[Some(89), Some(100)]); + assert_eq!(even.evidence.coverage(id), &[Some(50), Some(100)]); + // MUST NOT change: the reading set and the tag. + for t in [&swapped, &even] { + assert_eq!(set(t), set(&real)); + assert_eq!(t.pos("w"), real.pos("w")); + } + } + // ── D-LXC-11 coverage bands ── /// One two-state word per share: `n` carries `share` of 100, `v` the rest. @@ -1560,7 +1540,7 @@ mod tests { assert_eq!(band_of(70, c), CoverageBand::Decisive); } - // T7 the key is the dominant READING's share, never a state sum + // T7 the key is the most frequent READING's share, never a state sum #[test] fn the_key_is_reading_share() { let f = forms("1,x,n,9,40,w\n2,y,p,9,35,w\n3,z,v,9,25,w\n"); diff --git a/crates/deepnsm-v2/src/lexical.rs b/crates/deepnsm-v2/src/lexical.rs index fb2a30a5f..a1d143702 100644 --- a/crates/deepnsm-v2/src/lexical.rs +++ b/crates/deepnsm-v2/src/lexical.rs @@ -30,14 +30,18 @@ //! - **Not truth.** Counts are observed population evidence, stored as exact //! integers. No normalisation to `f32`, no probability, no NARS truth. //! -//! ## Frequency is the register's order +//! ## Frequency is evidence, not a lexical decision //! //! A word's readings are stored most frequent first, and each carries an -//! integer cumulative percentile coverage ([`LexicalEvidence::coverage`]). The -//! dominant reading is position 0 and its share is `coverage[0]` — a reader -//! never sums counts to find it. Source file order is not kept: COCA orders -//! `word_forms.csv` by lemma rank, so its first row is not the dominant reading -//! for 259 surfaces (`changes`: verb 13,624 first, noun 113,085 second). +//! integer cumulative percentile coverage ([`LexicalEvidence::coverage`]). +//! Position 0 is the most frequent observed reading and its share is +//! `coverage[0]` — a reader never re-sums counts to measure it. The order is a +//! presentation of the evidence, never a preference: every reading stays, and +//! nothing here chooses, ranks or eliminates one. Count changes may change the +//! order and the coverage; they never change which readings exist. Source file +//! order is not kept: COCA orders `word_forms.csv` by lemma rank, so its first +//! row is not the most frequent reading for 259 surfaces (`changes`: verb +//! 13,624 first, noun 113,085 second). //! //! ## Unknown is not zero //! @@ -264,8 +268,8 @@ impl LexicalEvidenceBuilder { /// Each word's readings are stored in FREQUENCY ORDER — highest known /// `form_count` first, unknown counts last, equal counts in insertion /// order — and each carries its cumulative percentile coverage (see - /// [`LexicalEvidence::coverage`]). Frequency is the register's order, so a - /// reader takes position 0 for the dominant reading and never re-sums. + /// [`LexicalEvidence::coverage`]). The order presents evidence; it selects + /// nothing, and every reading is kept. #[must_use] pub fn finish(mut self) -> LexicalEvidence { // Stable: equal keys keep insertion order. @@ -361,8 +365,8 @@ impl LexicalEvidence { } /// Every counted reading of word `id`, most frequent first (empty if none - /// survived, or `id` is out of range). Position 0 is the dominant reading - /// when [`coverage`](Self::coverage) is known. + /// survived, or `id` is out of range). Position 0 is the most frequent + /// observed reading; the slice is the reading set, not a ranked choice. #[must_use] pub fn readings(&self, id: WordId) -> &[LexicalReading] { &self.readings[self.span(id)] @@ -371,7 +375,7 @@ impl LexicalEvidence { /// Cumulative percentile coverage, aligned with [`readings`](Self::readings): /// entry `k` is the percent (`0..=100`, floored) of word `id`'s known /// occurrences covered by readings `0..=k`. The last entry is `100`, and - /// entry 0 is the dominant reading's own share. + /// entry 0 is the most frequent observed reading's own share. /// /// All `None` when any reading's count is unknown or every count is zero — /// a share of an unknown total is itself unknown. From a3ae189845162e6a0ce5baa2d84dbb49feddc5e6 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 30 Sep 2026 21:24:40 +0000 Subject: [PATCH 09/11] plan: mark the stale D-LXC-11 G8b result Claude-Session: https://claude.ai/code/session_01AUbvJZf4pD6GzkZZ6LtqSE --- .claude/plans/deepnsm-v2-coverage-bands-v1.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/.claude/plans/deepnsm-v2-coverage-bands-v1.md b/.claude/plans/deepnsm-v2-coverage-bands-v1.md index 4becfc46a..a94e2b757 100644 --- a/.claude/plans/deepnsm-v2-coverage-bands-v1.md +++ b/.claude/plans/deepnsm-v2-coverage-bands-v1.md @@ -216,6 +216,9 @@ agent. The entry carries a ready-to-paste STATUS_BOARD row for the operator: - G8a: 126 lib + 17 example tests pass; clippy `-D warnings` and fmt clean. - G8b: KJV unchanged — 70,396 triples, 1,237 subjects, G6 = 25. + ⊘ 2026-09-30: stale. After the #1304 repair the KJV run is 70,393 + triples, 1,227 subjects and 1,941 predicates, identical to `main`; G6 is + the count-change non-interference invariant. - G8c: population 141, cuts (72, 97), shares 50..99, bands 34/71/36; equal to the receipt. - Disable runs, each after the commit, each restored with `git checkout`: From da09899b6c175f2a274eff7efa25e61550b89f9a Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 30 Sep 2026 21:48:33 +0000 Subject: [PATCH 10/11] board: supersede the stale D-LXC-1 success claims The plan checklist and Results code block, and the AGENT_LOG entries, still presented the removed counted pick and its 25 moved tags as current. Mark them historical and point to the correction. Docs only. Claude-Session: https://claude.ai/code/session_01AUbvJZf4pD6GzkZZ6LtqSE --- .claude/board/AGENT_LOG.md | 9 +++++++++ .claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md | 8 ++++++++ 2 files changed, 17 insertions(+) diff --git a/.claude/board/AGENT_LOG.md b/.claude/board/AGENT_LOG.md index c41460051..dbf75b1f6 100644 --- a/.claude/board/AGENT_LOG.md +++ b/.claude/board/AGENT_LOG.md @@ -1,3 +1,12 @@ +## 2026-09-30 — correction to the two D-LXC-1 / D-LXC-11 entries below + +- Their "70,396 triples", "G6 = 25" and "25 tags moved (G6 exact)" record + frequency changing tags: unintended semantic interference, not a passing + gate. The #1304 repair removed the counted pick; the tag is `main`'s legacy + tagging again. KJV: 70,393 triples, 1,227 subjects, 1,941 predicates, + identical to `main`. G6 is now the invariant + `counts_change_evidence_never_the_readings_or_the_tag`. + ## 2026-09-30 — D-LXC-11 coverage bands (5+3 council, orchestrator implements) - The 5 (Sonnet): prior-art, iron-rule, code truth (general-purpose, diff --git a/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md b/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md index 8053cb649..bf44c3a93 100644 --- a/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md +++ b/.claude/plans/deepnsm-v2-lexical-evidence-consumer-v1.md @@ -158,6 +158,10 @@ Other facts read for this plan: ## Checklist (D-LXC-1; D-LXC-7 rides with it) +⊘ 2026-09-30: the counted-pick items below (per-`WordId` pick, resolution +order) record what was built and then removed; they are not the current +state. The current state is the Correction below. + - [x] Tests first (G3). Each was shown red by a disable run (see Results). - [x] In `bible_wave` only: build `LexicalEvidence` with `load_word_forms_csv` against `nsm.vocab`, after lowercasing the `word` column. @@ -179,6 +183,10 @@ Other facts read for this plan: ## Results (2026-09-29) +⊘ 2026-09-30: this Code block and the "after (B)" column describe the +removed counted pick; kept as the historical measurement. Current code and +numbers: the Correction below. + **Code.** `crates/deepnsm-v2/examples/bible_wave.rs`: - `counted_pos` implements the fold and pick. - `Tagger` is lemma table → counted pick → `archaic_pos` → Other. From b19a0455180d14d1491a98e936357ee208a85eab Mon Sep 17 00:00:00 2001 From: "coderabbitai[bot]" <136622811+coderabbitai[bot]@users.noreply.github.com> Date: Wed, 30 Sep 2026 21:56:15 +0000 Subject: [PATCH 11/11] docs(deepnsm-v2): Document lexical evidence span and out-of-range behavior --- crates/deepnsm-v2/src/lexical.rs | 2 ++ 1 file changed, 2 insertions(+) diff --git a/crates/deepnsm-v2/src/lexical.rs b/crates/deepnsm-v2/src/lexical.rs index a1d143702..26cc2710a 100644 --- a/crates/deepnsm-v2/src/lexical.rs +++ b/crates/deepnsm-v2/src/lexical.rs @@ -356,6 +356,8 @@ fn sum_known(counts: impl Iterator>) -> Result, E } impl LexicalEvidence { + /// The shared index range for word `id`'s readings and coverage entries. + /// Returns `0..0` when `id` is outside the stored offsets. fn span(&self, id: WordId) -> std::ops::Range { let i = id as usize; match (self.offsets.get(i), self.offsets.get(i + 1)) {