diff --git a/.prettierignore b/.prettierignore index dcf0642..b1f234d 100644 --- a/.prettierignore +++ b/.prettierignore @@ -7,3 +7,5 @@ package-lock.json docs/img .wrangler index.html +# V41 — generated from PLAN.md by `npm run changelog`; a test compares it byte for byte +CHANGELOG.md diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..07d4518 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,223 @@ +# Changelog + +Generated from the roadmap tables in `PLAN.md` by `npm run changelog` — not edited by +hand: a unit test fails when this file and the plan disagree, in either direction. One +entry per wave, newest first; the first six waves (V1–V6, the MVP) predate the tables and +are recorded in PLAN.md §B and §J. Each entry carries the opening sentence of its row and +the reason the wave was built. + +## V41 — The site says what it is + +Six corrections to what the site declared about itself — to crawlers, to link previews, to visitors — none of them touching the lab, all measured on production first (29/09/2026, `curl`). + +_Why:_ Owner request (29/09/2026): an analysis of the repository and of the production site, then a first wave. All six items are defects in what the site said about itself rather than missing features — a lab that publishes its refusals and its limits cannot have its documentation indexed as twelve copies of its home page, nor answer 200 to an address that does not exist. + +## V40 — Data Studio: validity, drift, and an auditable diff + +Quality was measured as completeness and consistency of type; what was missing is **validity** — a value can be present, correctly typed and still impossible. + +_Why:_ Came last because it builds on V38's faithful read and V39's per-column recipe: validity rules on mis-parsed numbers would have flagged the parser, not the data. Closes the Data Studio group. Of the 36 new unit tests, eleven assert that a rule REFUSES to fire — a rule that flags a good file is worse than no rule, because it teaches the reader to ignore the panel. + +## V39 — Data Studio: a recipe that works column by column + +`RecipeOptions` applied `missing` and `clipOutliers` to the **whole file** — one strategy for every column, however different they are. + +_Why:_ A single global strategy is the kind of default that looks tidy and quietly makes the data worse; per-column steps cost little to build because the recipe was already an object, not a pile of checkboxes. The build confirmed it: the engine change is contained in one stage of `applyRecipe`, and the 17 pre-existing recipe tests passed untouched. + +## V38 — Data Studio: reading the file exactly as it was written + +The headline item was a **defect in shipped code, not a missing feature**, and the wave opened by proving it. + +_Why:_ Owner request (22/08/2026): what to improve in /data. The audit found a defect first, and the wave began by reproducing it end to end: a French-locale CSV — the single most likely file this owner's users will open — silently loses every numeric column. The repair is exact rather than approximate: the French file now trains to the same types, the same feature count and the same score, to ten decimal places, as the file that never had the problem. + +## V37 — ML Lab: speed and the comfort of long sessions + +The wave opened, as the plan demanded, with a measurement — and the measurement moved the wave. + +_Why:_ Launched 23/08/2026, after V36. The plan said « measure before and after so the gain is published rather than claimed » — and the measurement is what turned the wave around: the promised parallelism was worth 6%, while the bottleneck it revealed was worth 5×. Two latent defects fell out of the same instrumentation: models corrupted by structured clone, and an inference column reading 0 ms for every parallel family. + +## V36 — ML Lab: the gaps that were deliberately left open + +Each item was consciously deferred in an earlier wave rather than forgotten; delivering them together keeps the descopes visible instead of letting them quietly become permanent. + +_Why:_ Launched 23/08/2026, right after V35. Each item was a named descope, not an oversight — and the ensemble exposed one more silent-disappearance defect, of the same family as the one V35 found. + +## V35 — ML Lab: the number stops flattering itself + +Two method defects in shipped code, fixed, plus the two additions that follow from them. + +_Why:_ Owner request (22/08/2026), launched 23/08/2026. Two of the four items were defects rather than gaps: a lab that sells honest evaluation cannot ship a headline figure it knows to be optimistic, nor a split that leaks on dated data. + +## V34 — The explanations, the how-to guides, and a limits page extracted from this very file + +Six new pages per language — two explanations (the method choices; what LabML does not do) and four task-shaped how-to guides (score a batch, compare two runs, read a learning curve, hand a SQL result to the lab) — bringing the documentation to **twelve pages per language across all four Diátaxis quadrants**. + +_Why:_ Three audiences, deliberately: the curious visitor (five minutes), the practitioner (one task), and the evaluator judging whether the engineering is rigorous. The explanation pages are what the third one reads. + +## V33 — The reference, and a table of refusals extracted from the code rather than from memory + +Five pages per language — the refusals table, ML Lab, Data Studio, Vision & assistant, and file formats — grouped by section rather than one page per panel, so a lookup lands on one page with anchors instead of hunting across twenty. + +_Why:_ A feature nobody can look up is a feature that does not exist for the reader; and a refusal nobody can decode reads as a bug rather than as the design it is. + +## V32 — Documentation that cannot lie — and a measurement that rewrote this row's own rule + +A `/docs` route, linked from the footer, built on the **Diátaxis** split, with the Markdown living in `src/content/docs//*.md` and compiled **at build time**: the reader downloads finished pages, an outline and a search index — never a parser, and never a request to a documentation host. + +_Why:_ Owner request (22/08/2026): document every shipped feature across /ml, /data and /ai, linked from the footer. One finished tutorial first, on purpose — writing the full reference before the template is settled means rewriting all of it. + +## V31 — Vision that says « I do not know » — and a bench that refuted three of this row's own predictions + +**(A) Measure first**, as V30 taught: the complaint « it still makes mistakes » is not a measurable statement. + +_Why:_ Owner report (22/08/2026): the vision playground is better than the chat but still makes mistakes. Naming the label-space mismatch is what turns a vague complaint into a fixable defect. + +## V30 — Chat that reads better, measured before it is made bigger + +The wave began by building the instrument, because V27's stood on 18 cases that needed a GPU with `shader-f16` — one laptop's worth of evidence, re-runnable by nobody. + +_Why:_ Owner question (22/08/2026): would a bigger model raise the share of correct answers? The wave answers with a measurement rather than an estimate, and the answer has two halves. For **0 MB**, the app went from **33 right / 15 wrong** to **42 right / 7 wrong** out of 55 — nine more correct answers and **fifty-three percent fewer wrong ones**, the single largest piece of which came from the deterministic parser rather than the model. And the second half was measured too, not deferred: **Qwen3-1.7B at 1.43 GB — four times the download — scores worse** (40 right / 12 wrong against 42 / 7). It reads the hard questions better and the easy ones worse. « Bigger » is not a direction of improvement on this task; it is a trade whose sign has to be measured, and the bench now measures it in one command. + +## V29 — Analytical SQL in the browser (DuckDB-Wasm, MIT) + +**Analytical SQL in the browser (DuckDB-Wasm, MIT)**: the Data Studio gains a real OLAP engine — joins, window functions, aggregations — over the file you just loaded, with no server and no upload. + +_Why:_ Owner request (21/08/2026): real analytical SQL on ~100 MB files with zero backend. Delivered after the /privacy page at the owner's request (22/08/2026). + +## V28 — « Ne nous croyez pas sur parole » + +**« Ne nous croyez pas sur parole »** — a `/privacy` route that states the local-only promise once, in full, and then hands the reader the means to check it without trusting a word of it. + +_Why:_ Owner request (22/08/2026): the promise is repeated across the site, but a user has no way to tell a true claim from a comforting one. Verifiability is the product here — anyone can write « your data stays local » in a footer. + +## V27.3 — A `>=` that was quietly an `=` + +**A `>=` that was quietly an `=`**: retesting the comparison question after V27.2 produced « 0 ligne correspond où fare >= 0 » — impossible on a table where all 891 fares clear zero. + +_Why:_ Found by retesting in production (22/08/2026). An arithmetically impossible answer — zero rows for a condition every row satisfies — is worse than a refusal and worse than a wrong reading: it makes the engine itself untrustworthy, which is the one thing LabML sells. + +## V27.2 — Two honesty defects, one measured, one found while reading the measurement + +**Two honesty defects, one measured, one found while reading the measurement**: (1) the comparison question V27.1 left wrong — « est-ce que les femmes payaient plus cher que les hommes ? » read as a correlation between `fare` and `age` — gets a rule that names both halves of the mistake: a question comparing two groups is an aggregate with `groupBy` on the column whose values name them, NEVER a correlation; and never pick a column the question does not mention. + +_Why:_ Measured by the owner on real hardware (22/08/2026): 5 of 6 reference questions right after V27.1. The sixth is a confidently wrong answer to a different question than the one asked, and the « sur 891 lignes » wording was found by reading that same screenshot closely — a right number inside a wrong sentence is exactly what this project refuses to ship. + +## V27.1 — The model earns its place, it does not take it + +**The model earns its place, it does not take it**: the V27 order was wrong, and the measurement said so. + +_Why:_ Measured by the owner in production (22/08/2026), the day V27 shipped. A confidently wrong answer costs more trust than a refusal — and V27 produced two of them, including a 0 where the deterministic engine already had the right 168. + +## V27 — Local chat, upgraded + +**Local chat, upgraded**: a real language model — **Qwen3-0.6B-DQ, 355 MB, Apache-2.0** — running entirely in the browser, offered beside the V6 deterministic interpreter, which stays the DEFAULT and the fallback. + +## V26 — Learning curves + +**Learning curves**: the lab answers the classic budget question — "would more data help this model, or is it time to work on features?" — with one new chart. + +## V25 — Scale + +**Scale**: the lab now takes 100k–1M-row files without dying, on a measure-first design. + +## V24 — Text columns + +**Text columns**: free text stops being skipped and enters the pipeline as a hand-written **TF-IDF** block — accent-folding bilingual tokenizer, merged FR/EN stop words, vocabulary capped at 256 terms ranked by document frequency (ties alphabetical, terms seen in a single training document dropped), smoothed IDF, L2-normalized vectors, fitted on the training split only. + +_Why:_ Real CSVs have text columns (comments, descriptions) — the lab used to drop them on the floor + +## V23 — Vision 2 + +**Vision 2**: SqueezeNet (2012) retired for three self-hosted ONNX models — **EfficientNet-Lite4 int8** classification (1,000 ImageNet classes, 77.6% top-1), **YOLOX-Nano** object detection (80 COCO classes; the stronger-but-AGPL YOLOs were ruled out, Apache-2.0 kept) and **UltraFace RFB-320** face detection — boxes drawn on the image, FR/EN class names, plain-language counts ("1 person · 1 face"). + +_Why:_ Owner request (21/08/2026): portraits have no ImageNet class, so the old model answered off-target — and the detector must recognize a whole range of things, not just faces + +## V22 — The model comes back + +**The model comes back**: export as format v2 — the JSON embeds the fitted pipeline (imputation/encoding/standardization), the target, the classes and the exporting run's test metrics as an honest reference. + +_Why:_ Closes the last loop: train today, come back in a month, score + +## V21 — Compare two runs + +**Compare two runs**: check two runs in the history → side-by-side diff on /ml/compare — features added/removed as ± badges, every model's metric in an A/B/Δ table (signed colors), plain-language read of the best model's movement, and a cross-run verdict when both runs carry V20 CIs (disjoint → the gap exceeds both uncertainties; overlapping → possibly noise). + +_Why:_ "Did my cleaning help?" — the central iterative gesture of ML + +## V20 — Honest uncertainty + +**Honest uncertainty**: seeded bootstrap of the test set (1,000 resamples shared across models — paired comparisons) → percentile 95% CI on every leaderboard model's main metric (whiskers on a shared scale), plain-language paired winner-vs-baseline verdict ("the gap survives resampling — probably real" / "the interval crosses zero — possibly noise"), analysis attached to the run (history/report/share). + +_Why:_ `0.82` on 178 rows is not `0.82`; say what the number does not say + +## V19 — Persistent projects + +**Persistent projects**: the dataset joins the project, opt-in ("keep in this browser") — lz-string-compressed CSV in IndexedDB, explicit 50 MB budget (named refusal with the numbers, never a silent cut), saved list (reopen/forget) under the history, runs linked to the saved dataset ("reopen this run's data"), identical retraining (seed 42). + +_Why:_ A refresh erased everything; "projects" are only real if they survive + +## V18 — Per-segment analysis + +**Per-segment analysis**: after a run, the test set is sliced by every categorical column — including those excluded from the features, where proxy effects hide — and the inspected model's metric (accuracy or RMSE) is recomputed per slice, gap vs global signed and sorted worst-first. + +_Why:_ "Where does my model fail?" — an honest gateway to fairness + +## V17 — Data Studio 3: joins & anomalies + +**Data Studio 3: joins & anomalies**: left join of a second file on a shared key (exact match after trim — a dirty key becomes a named orphan, never silence; match rate, duplicates, unused rows; the joined result becomes THE dataset), and **multivariate anomalies** via a hand-written seeded isolation forest (100 trees, exact c(n)) as a step of the **replayable recipe** (threshold 0.6). + +_Why:_ Real data prep starts by crossing two files; multivariate anomalies see what Tukey misses + +## V16 — Imbalance & thresholds + +**Imbalance & thresholds**: precision-recall curve (AP, chance line drawn), adjustable decision threshold priced by a cost matrix (false alarm vs missed case, one-click optimum), calibration curve (Brier), imbalanced demo `fraud.csv`; the chosen threshold joins the run. + +_Why:_ Real datasets are imbalanced; accuracy lies there + +## V15 — Score a new batch + +**Score a new batch**: after a run, drop a new file → the inspected model scores it in the browser (exportable predictions, all columns preserved); if the target is present, honest test-vs-batch comparison (unknown labels excluded and counted); schema validated, drifted demo `iris-field.csv`, score attached to the run (history/report/share) + +_Why:_ The complete MLOps loop: V11 says "the inputs moved", V15 says "does the model still hold" + +## V14 — Generalized prerendering + +**Generalized prerendering**: static shells for all six sections (the V9 approach extended — inlined CSS, Latin fonts as data:, per-route preloaded façade, header template), Lighthouse /ml 0.86 → 0.99 and /data 1.0 under real throttling (3-run medians); the root stays the SPA fallback (accepted) + +_Why:_ The last Lighthouse gap + +## V13 — Complete runs + +**Complete runs**: tuning, latest Shapley explanation, exploration and forecast attached to the run record — IndexedDB history (with chips), stored-run page, HTML report and v2 share links (subsampled scatter plots in the URL; v1 links remain decodable) + +_Why:_ The V5–V8 artifacts did not survive the run + +## V11 — Data drift + +**Data drift** in the Data Studio: a reference file, a file to compare → schema differences (columns added/removed/retyped), **PSI per column** (quantile bins from the reference, thresholds 0.1/0.25), new/vanished categories, missing-rate gaps, overall verdict — with a deliberately drifted demo (`cafe-sales-june.csv`) + +_Why:_ The MLOps gesture par excellence: checking that a new batch looks like what the model learned on + +## V10 — Data Studio 2 + +**Data Studio 2**: importable recipe replayable on a new file, per-column forced types, derived columns + +_Why:_ Completes the reproducibility loop + +## V9 — Performance & comfort + +**Performance & comfort**: /ml Lighthouse budget ≥ 0.90 (preloads, splitting), PWA update toast, webcam for vision (Permissions-Policy to open) + +_Why:_ Perceived quality and scores + +## V8 — Time series + +**Time series**: date + numeric target detection → trend/season decomposition, hand-written Holt-Winters forecasting, rolling-origin backtest + +_Why:_ Opens up an entire class of problems + +## V7 — Unsupervised exploration + +**Unsupervised exploration** in the ML Lab: hand-written k-means (seeded k-means++ init, k ∈ 2–5 chosen by silhouette), hand-written 2D PCA projection (power iteration), plain-language group profiles, scatter plot with colors **and shapes** (palette validated for color blindness by the design-system validator) + +_Why:_ Fills the real gap: today the lab requires a target; many datasets are explored first without one diff --git a/PLAN.md b/PLAN.md index 1ca0416..418b207 100644 --- a/PLAN.md +++ b/PLAN.md @@ -442,7 +442,7 @@ data budget asks. | **V29 — delivered** | **Analytical SQL in the browser (DuckDB-Wasm, MIT)**: the Data Studio gains a real OLAP engine — joins, window functions, aggregations — over the file you just loaded, with no server and no upload. The file is queried **as dropped, before the cleaning recipe**: the recipe belongs to the studio, and a result traceable to nothing the user can reopen would be worse than no SQL at all. Extra CSV / **Parquet** / JSON files can be attached in the same session (Parquet is a new input format for the lab), each exposed as a view named after the file; a result exports to CSV or goes to the ML Lab in one click, through the handoff path V4 already built. Errors show **DuckDB's own message** — it names the line and the token, which no paraphrase of ours would. **The measurement that set the version**: `@duckdb/duckdb-wasm` is pinned to **1.28.0**, not `latest`. From 1.29 the binaries cross Cloudflare Pages' hard 25 MiB per-file limit (eh 34.2 MiB, mvp 39.4 MiB); at 1.28.0 they are **17.3 and 21.1 MiB** and fit. Newer would have meant sharding the wasm and either widening `connect-src` to `blob:` — days after publishing a page that quotes that very directive — or rebuilding the service worker in injectManifest mode. An older engine was the cheaper honest trade, and it is written here so the next upgrade re-measures instead of rediscovering. Self-hosted under `/duckdb/` (the library defaults to jsDelivr, which the CSP refuses), **never precached** — cached on first use like the vision models, so nobody pays 18 MiB before opening the console — and the `coi` threaded build is left out entirely: no COOP/COEP, no SharedArrayBuffer, single-threaded as the assumed mode. Remote S3/HTTP querying stays out, by CSP and by intent. 352 unit tests, 61 e2e. | Owner request (21/08/2026): real analytical SQL on ~100 MB files with zero backend. Delivered after the /privacy page at the owner's request (22/08/2026). | | **V30 — delivered** | **Chat that reads better, measured before it is made bigger.** The wave began by building the instrument, because V27's stood on 18 cases that needed a GPU with `shader-f16` — one laptop's worth of evidence, re-runnable by nobody. It now stands on **55 reference questions**, French and English, over every shape of the query grammar plus three that no query can answer, where refusing is the only correct outcome. Two harnesses run it: one in CI on every commit with no model at all (the deterministic parser, the grammar automaton, the token mask), and one against the REAL pinned q4f16 weights on a CPU through onnxruntime-node — same files, same prompt, same decoding path as production, minus the GPU. « Measurable » stopped meaning « on one machine ». **The instrument immediately contradicted the wave's own premise.** The failure was not mainly the model: on those 55 questions the shipped app answered **33 right, 15 WRONG, 7 refused** — and **seven of the fifteen wrong came from the deterministic parser**, which runs first and can never be overridden. « Combien de femmes ? » answered 891 instead of 314: the grammar knows `combien`, knows nothing about `femmes`, kept the count and dropped the condition — under the badge that is supposed to mean exact. Four of those seven the local model reads correctly, and never got asked. **So the parser now checks its own coverage**: every word of the question must be accounted for by a lexicon phrase, a column the answer uses, a value it filters on, or one of three closed lists (the table's own furniture, generic row nouns, grammatical filler). A leftover word is a refusal. Measured: **19 right, 0 wrong, 36 refused** — the seven wrong answers became refusals and not one correct answer was lost. `wrong === 0` is now asserted in CI, and the trade is one-directional by construction: an unknown word can cost a refusal where an answer was possible, never a wrong answer where a refusal was right. **(B) Constrained decoding**, hand-written: an automaton over the query grammar and a `LogitsProcessor` that masks, at every token, everything that would leave it. It walks UTF-8 **bytes**, not characters, because Qwen's vocabulary is byte-level BPE and 1 457 of its 151 669 tokens are fragments of a character — a character-level automaton would have made « Île-de-France » unwritable as a filter value. It cost about 16 ms at its most expensive step (`{"kind":"` masks seven letters against seven large buckets) after the first-byte step was hoisted out of the per-token loop, down from 53 ms. **And on its own it made things worse**: model refusals fell 14 → 2 and correct answers rose 29 → 34, but **wrong answers rose 12 → 19**. Forcing a valid answer turns « I could not parse that » into a confident wrong number. That is why the grammar keeps `{"kind":"none"}` reachable — a shape whose only meaning is « I cannot express this », which maps to the refusal V27 already had. **(C) Examples drawn from the user's own columns**, for 0 MB — and the reason turned out to be sharper than the plan's. V27's nine examples were frozen Titanic, and **seven of the 55 corpus questions appear in them verbatim**: the prompt had been fitted to the bench across V27.1 and V27.2, so on those questions the old bench could not tell reading from recitation. (Checked rather than assumed: on those seven, before and after score identically, 5 right / 1 wrong / 1 refused. The defect is methodological, and its measured effect on this comparison is zero.) Generating the examples from the loaded file removes the contamination structurally and deletes the rule that asked the model to ignore what it had just been shown. **Two versions of them were worse than the frozen ones, and both reasons are now in the code.** The first left out the aggregate-WITH-FILTER shape: under constraint, a shape the model has not been shown comes out as a confident wrong answer rather than a refusal, and « prix moyen payé par les survivants » became `count where fare = 1000000000`. The second still picked the FIRST numeric column, which on Titanic is `survived` — so the examples read « average survived » and the model duly reached for that column on questions that never mention it. `ColumnInfo` gained a capped `distinct` count so an example averages a **quantity**, not a 0/1 flag. **(D) Two samples, one vote — dropped, on this wave's own measurements.** Both halves of its tie-break died with (B): constrained decoding guarantees every candidate validates, so « keep the one that validates » no longer discriminates; and « invents no column the question never names » is contradicted by the corpus, where « did women pay more than men? » is correctly answered with `fare`, a column the question never names. Two samples for a vote with no criterion, at twice the latency, is not a trade. **What the wave deliberately does not do**: ship a second, bigger model as a download (the cheap levers were not exhausted when the plan proposed it, and now they are), guess a column by fuzzy name-matching (the refusal is the honest outcome), or let the grammar automaton replace `validateIntent` — it over-approximates in two named places and is a filter, not the authority. 557 unit tests, 79 e2e. | Owner question (22/08/2026): would a bigger model raise the share of correct answers? The wave answers with a measurement rather than an estimate, and the answer has two halves. For **0 MB**, the app went from **33 right / 15 wrong** to **42 right / 7 wrong** out of 55 — nine more correct answers and **fifty-three percent fewer wrong ones**, the single largest piece of which came from the deterministic parser rather than the model. And the second half was measured too, not deferred: **Qwen3-1.7B at 1.43 GB — four times the download — scores worse** (40 right / 12 wrong against 42 / 7). It reads the hard questions better and the easy ones worse. « Bigger » is not a direction of improvement on this task; it is a trade whose sign has to be measured, and the bench now measures it in one command. | | **V31 — delivered** | **Vision that says « I do not know » — and a bench that refuted three of this row's own predictions.** **(A) Measure first**, as V30 taught: the complaint « it still makes mistakes » is not a measurable statement. The bench is **14 images, not the 30–50 the plan asked for**, and the shortfall is deliberate: a search-and-download pipeline produced a backlit dog, a triptych of broccoli close-ups and an eighteenth-century painting of a lighthouse, each of which would have scored the model wrong for the corpus's own mistakes. **Every kept image was looked at**, and the ones whose ground truth could not be stated were thrown away — fewer and verified is an instrument, more and unverified is noise with a percentage attached. Licences are CC0, public domain or CC BY only, never CC BY-SA, whose share-alike term would propagate into an MIT repository. It runs in the ordinary e2e suite, in the real browser, through the real canvas and the real worker, in about 15 seconds. **What it found**: where an ImageNet label exists at all the classifier is **9/10 top-1**; the object detector is **7/9**; face counts are **10/10 exact**. And on the four images no label in the thousand can name, the page answered anyway — « stage » at 36.7 %, « golf cart » at 37.1 %, « semi-trailer truck » at 41.2 %, « football helmet » at **86.6 %**. **(C) The honest refusal, for 0 MB.** A softmax over 1000 classes cannot abstain — it renormalises to 1 whatever it is shown — so the refusal is decided outside the classifier, by two rules that were each chosen against the measurement rather than picked and hoped for. **Rule 1, the human subject**: fire when the object detector finds a person AND the face detector finds a face. Requiring both is what the bench bought — YOLOX drew a `person` box on a wine bottle and on a red sports car, and UltraFace found no face on either, so a person box alone would have cost two correct answers. **Rule 2, a confidence floor at 50 %**: on this corpus the cheapest correct answer is 70.5 % and the dearest wrong one not already taken by rule 1 is 41.2 %. Rule 1 wins over rule 2, because « football helmet » at 86.6 % walks straight through any floor. Measured: **4/4 unnameable images refused, 1/1 wrong answer announced as wrong, 0/9 correct answers lost** — both guarantees frozen as assertions in the bench. The label is never hidden, only framed: the top-5 stays on screen with its probabilities, because a number nobody can see is a number nobody can check. **Three of this row's own predictions were wrong, and the measurement said so.** (i) « Squashing a 16:9 photo into a square skews everything — `preprocess.ts` is the suspect »: the code never squashed, it centre-crops, and replacing the crop with a full-frame squash scores **identically, 9/10, with the same single wrong answer**. The crop is not the defect. (ii) Averaging two crops therefore has nothing to buy — both views agree on the failure. (iii) Recalibrating `OBJECT_THRESHOLD`: probed at 0.05, YOLOX-Nano does not see the football at **any** threshold (its best guesses on that image are « toilet » 14 % and « toaster » 8 %) and does not see the bottle either. Lowering the bar recovers neither miss and admits noise; the V23 value stands. The squash experiment also produced the strongest argument for rule 1 existing at all: under it, the photograph of people under an umbrella comes back as « umbrella » at **99.3 %** — plausible, confident, and not what the picture is of. No confidence floor would ever catch that one. **(B) CLIP ViT-B/32 (~190 MB) — not in this wave**, and for a reason the bench can state precisely: the defect this wave targeted is fully covered for 0 MB, and CLIP would answer a **different** question. The refusal turns a false answer into an honest silence; it does not turn it into a right answer. Whether 190 MB is worth turning that silence into a name is a wave-scale decision, and the bench cannot arbitrate it, because it scores a task CLIP would replace rather than improve. **(D) YOLOX-S (~35 MB) — not in this wave either, and this one the bench does arbitrate**: the two object misses are inert. Both images are already named correctly by the classifier, and both of the detector's false `person` boxes are already neutralised by rule 1's face requirement, at no cost. That is +31 MB, plus the < 25 MiB splitting machinery V27 needed for Cloudflare Pages, to buy two chips on two images and nothing at all for the verdict. **What this wave deliberately does not do**: use the detector to correct the classifier (« the detector says cat, so the label `French Bulldog` is wrong » is a COCO-to-ImageNet taxonomy problem, and that image is already caught by the floor — a third rule with zero measured effect is an untested parameter, which is exactly what V30 warned against), hide the label behind the verdict, or present the 50 % floor as a law: **it is estimated from fourteen images**, the gap it sits in is wide but weakly sampled, and it is written down as a named constant so the next corpus can move it. 566 unit tests, 80 e2e. | Owner report (22/08/2026): the vision playground is better than the chat but still makes mistakes. Naming the label-space mismatch is what turns a vague complaint into a fixable defect. | -| **V32 — delivered** | **Documentation that cannot lie — and a measurement that rewrote this row's own rule.** A `/docs` route, linked from the footer, built on the **Diátaxis** split, with the Markdown living in `src/content/docs//*.md` and compiled **at build time**: the reader downloads finished pages, an outline and a search index — never a parser, and never a request to a documentation host. Search is local for the same reason the models are self-hosted: a site whose whole claim is that nothing leaves cannot make its own docs the exception. At this corpus size the index is a scan, stated as such rather than dressed up. **The wave began by measuring, and the measurement is what made rule (1) buildable.** « The docs are tested like the code » only works if you know which numbers are stable, so the tutorial's exact path was run twice before a word was written: **every metric came back identical, and every wall clock did not** — Random forest trained in 12 050 ms and 13 231 ms on the same machine, minutes apart. A page quoting « trained in 0.7 ms » would therefore break the build with nothing broken. That is now its own e2e test: **no documentation page may quote a duration**, in any language, ever. The measurement also corrected this row: the plan illustrated « you will get 0.821 accuracy » and the real champion figure is **0.792**. **The tutorial teaches the thing the app already does honestly**: on titanic the elected champion is k-nearest neighbors at 0.818 on validation, and it is only **sixth on test at 0.792** — the decision tree scores 0.831. A tutorial that printed 0.831 in large type would be lying about its own method, so the page makes the −0.026 selection gap its central lesson instead of a footnote. **The guard was wrong twice before it was right, and both failures were found by attacking it.** Version one asserted the page CONTAINS the right figures — so editing one occurrence of 0.792 into 0.800 still passed, because the other occurrence kept the assertion true. Version two read the figures OUT of the page and demanded the app produce each one, but scoped that to blockquotes and tables — so the same edit in **bold prose** still passed, and bold prose is where the reader actually takes the number from. Version three covers quotes, tables and bold, and was re-attacked on all three: a wrong figure in English prose, a wrong figure in a French table, and a quoted duration each fail now. Its limit is stated in the spec rather than left to be discovered: **plain prose is not checked**, because prose also carries numbers that are reasoning rather than quotation (« ten minutes », « 62% of the rows »), so the convention is that a figure you want checked goes in a quote, a table, or bold. **(3) Better than a screenshot, a link that does the thing**: `:::try /ml?demo=titanic | label`compiles to a deep link, and`/ml`now reads`?demo=`and`?target=`. A screenshot is a claim about the past that rots silently; a deep link either works or the e2e catches it. The parameter becomes a fetched path, so it is resolved against the shipped demo list and nothing else — an e2e test asserts that `?demo=../../../etc/passwd`fetches nothing. **What this wave deliberately does not do**: write the reference pages before the template is settled (rewriting all of them is the predictable cost), hand-take a screenshot, or claim the figure guard is total.`marked` is a build-time devDependency and never reaches the browser. 601 unit tests, 86 e2e. | Owner request (22/08/2026): document every shipped feature across /ml, /data and /ai, linked from the footer. One finished tutorial first, on purpose — writing the full reference before the template is settled means rewriting all of it. | +| **V32 — delivered** | **Documentation that cannot lie — and a measurement that rewrote this row's own rule.** A `/docs` route, linked from the footer, built on the **Diátaxis** split, with the Markdown living in `src/content/docs//*.md` and compiled **at build time**: the reader downloads finished pages, an outline and a search index — never a parser, and never a request to a documentation host. Search is local for the same reason the models are self-hosted: a site whose whole claim is that nothing leaves cannot make its own docs the exception. At this corpus size the index is a scan, stated as such rather than dressed up. **The wave began by measuring, and the measurement is what made rule (1) buildable.** « The docs are tested like the code » only works if you know which numbers are stable, so the tutorial's exact path was run twice before a word was written: **every metric came back identical, and every wall clock did not** — Random forest trained in 12 050 ms and 13 231 ms on the same machine, minutes apart. A page quoting « trained in 0.7 ms » would therefore break the build with nothing broken. That is now its own e2e test: **no documentation page may quote a duration**, in any language, ever. The measurement also corrected this row: the plan illustrated « you will get 0.821 accuracy » and the real champion figure is **0.792**. **The tutorial teaches the thing the app already does honestly**: on titanic the elected champion is k-nearest neighbors at 0.818 on validation, and it is only **sixth on test at 0.792** — the decision tree scores 0.831. A tutorial that printed 0.831 in large type would be lying about its own method, so the page makes the −0.026 selection gap its central lesson instead of a footnote. **The guard was wrong twice before it was right, and both failures were found by attacking it.** Version one asserted the page CONTAINS the right figures — so editing one occurrence of 0.792 into 0.800 still passed, because the other occurrence kept the assertion true. Version two read the figures OUT of the page and demanded the app produce each one, but scoped that to blockquotes and tables — so the same edit in **bold prose** still passed, and bold prose is where the reader actually takes the number from. Version three covers quotes, tables and bold, and was re-attacked on all three: a wrong figure in English prose, a wrong figure in a French table, and a quoted duration each fail now. Its limit is stated in the spec rather than left to be discovered: **plain prose is not checked**, because prose also carries numbers that are reasoning rather than quotation (« ten minutes », « 62% of the rows »), so the convention is that a figure you want checked goes in a quote, a table, or bold. **(3) Better than a screenshot, a link that does the thing**: `:::try /ml?demo=titanic \| label` compiles to a deep link, and `/ml` now reads `?demo=` and `?target=`. A screenshot is a claim about the past that rots silently; a deep link either works or the e2e catches it. The parameter becomes a fetched path, so it is resolved against the shipped demo list and nothing else — an e2e test asserts that `?demo=../../../etc/passwd`fetches nothing. **What this wave deliberately does not do**: write the reference pages before the template is settled (rewriting all of them is the predictable cost), hand-take a screenshot, or claim the figure guard is total.`marked` is a build-time devDependency and never reaches the browser. 601 unit tests, 86 e2e. | Owner request (22/08/2026): document every shipped feature across /ml, /data and /ai, linked from the footer. One finished tutorial first, on purpose — writing the full reference before the template is settled means rewriting all of it. | | **V33 — delivered** | **The reference, and a table of refusals extracted from the code rather than from memory.** Five pages per language — the refusals table, ML Lab, Data Studio, Vision & assistant, and file formats — grouped by section rather than one page per panel, so a lookup lands on one page with anchors instead of hunting across twenty. **The wave began by inventorying, not by writing.** A refusal table written from memory documents the memory; this one was extracted from the source: **38 kebab codes are thrown, 2 of them only by tests** (`constrained-decoding-unavailable` in the LLM bench, `header-not-served` in the privacy test) and are therefore absent from the public table, because a row nobody can encounter is padding. The four codes this row names all exist. `refusals.ts` lists every shipped code with its audience, and **`refusals.test.ts` re-extracts them on every run in both directions**: a code thrown but unlisted fails, a code listed but no longer thrown fails. Both were proved by breaking them on purpose. Two further tests tie the catalogue to the pages — every visitor-facing code must appear in both languages, and no page may name a code the code does not have. **The inventory found a real defect, which is what an inventory is for.** `dataset-missing` — a kept dataset that IndexedDB no longer holds — had no message of its own and fell through to « the file looks empty or not tabular. Try a CSV with a header row », sending the reader to hunt a format problem that does not exist. Worse than an undecodable refusal: a **mis-decoded** one. It now says what actually happened. It also cleared two false suspects: `not-parallelisable` and `not-serialisable` are not refusals at all — `parallel-run.ts` drops them and the main worker trains that family sequentially, same result, only slower — so they are documented as guards that are deliberately silent rather than listed as failures. **V32's figure guard had to be narrowed, and the measurement is why.** It asserted that every figure quoted in any doc page appears in a live titanic run. That is right for a tutorial, which reproduces a run, and meaningless for a reference page quoting a 0.35 threshold or a 13.6 MB model. Run against the new pages it went green **while genuinely checking nothing** — it passed on coincidental small integers. It is now scoped by the front matter's Diátaxis `kind`, and the reference pages get their own guard: named constants are checked against the file that defines them, and the three vision model sizes against the bytes on disk. That guard **caught a wrong figure in its own first run** — the reference page said the three models weigh 18.6 MB, copied from this very row; they weigh **18.5**. Its bound is stated in the spec: it checks the constants listed, not every number on a page. **What this wave deliberately does not do**: one page per panel (twenty pages make lookup worse, not better), generate the reference from the code (it produces mush, as this row predicted), or claim the guards are total. 611 unit tests, 88 e2e. | A feature nobody can look up is a feature that does not exist for the reader; and a refusal nobody can decode reads as a bug rather than as the design it is. | | **V34 — delivered** | **The explanations, the how-to guides, and a limits page extracted from this very file.** Six new pages per language — two explanations (the method choices; what LabML does not do) and four task-shaped how-to guides (score a batch, compare two runs, read a learning curve, hand a SQL result to the lab) — bringing the documentation to **twelve pages per language across all four Diátaxis quadrants**. **The limits page is extracted, not remembered**, the way V33's refusals were: `PLAN.md` records what every wave deliberately refused, and the page reads it back. That matters because memory flatters — elegant renunciations are easy to recall and embarrassing ones are not. The page therefore separates **three kinds of limit** that are usually conflated: a design choice (feasible, but it would have cost something better), a **measured drop** (built or costed, and the measurement said no), and a **refuted prediction** (the plan asserted, the measurement disagreed, and measurement won). The third section is the uncomfortable one and the most useful: it lists this file's own errors — the crop that was never a squash, « you will get 0.821 » against a real 0.792, 18.6 MB against a real 18.5, V27's inverted interpreter order. A guard traces each quoted figure back to the PLAN entry that recorded it, so the page cannot keep asserting a renunciation that was later reversed — **which has already happened once**: class weighting was descoped by name in V16 and delivered in V36, and the page says so, because that is what separates « not done » from « forgotten ». **« No next step » is now a test, not a habit.** Every page must end with « Et ensuite ? » / « Where to go next » AND offer at least one working link; the guard found **ten of the twelve existing pages** had no such section and the English tutorial had one with no links at all. Three more page-level guards came with it: every slug exists in both languages (a slug in one language only sends a language-switching reader to a « page does not exist »), every `/docs/…` link resolves to a real slug, and all four quadrants are represented. **The guard's own bug was caught by the guard.** It demanded the figure « 1.43 » in both languages and failed on the French page, which correctly writes « 1,43 » — a defect in the check, not in the page; it now accepts either decimal separator, and only that. **What this wave deliberately does not do**: repeat `/privacy` in prose (two texts on one subject diverge, and it is the forgotten one that survives — the method page links instead), ship hand-taken screenshots, or extrapolate a learning curve beyond what was measured. 672 unit tests, 88 e2e. | Three audiences, deliberately: the curious visitor (five minutes), the practitioner (one task), and the evaluator judging whether the engineering is rigorous. The explanation pages are what the third one reads. | | **V35 — delivered** | **ML Lab: the number stops flattering itself.** Two method defects in shipped code, fixed, plus the two additions that follow from them. **(1) The winner was picked on the test set** — the leaderboard sorted nine models by the metric computed on test and crowned `sorted[0]`, which makes the headline figure the optimistic maximum of nine draws. There is now a **third split**: validation is carved out of the train side (64/16/20 with the default ratios), ranking and crowning happen on validation, and the champion line spells out both numbers and the gap between them — « selected on validation at 0.974, scores 0.917 on the untouched test set ». The **test indices are byte-identical** to what the same config produced before V35, so every panel that reads the test set (segments, thresholds, uncertainty, batch compare) is unchanged; below 60 usable rows the third split is refused by name and the lab ranks on test as before. Ranking now lives in ONE module (`ranking.ts`) used by the leaderboard, the history, the run comparison, the report and the auto-selected insights model — the bug that shipped mid-wave was exactly that a fourth site still sorted on test and opened a different model than the one crowned. **(2) The split was always random, even on dated data.** A chronological split (oldest rows train, newest test, rows without a parseable date dropped and counted) and a group split (no group on both sides — the same customer in train and test is the same leak) are offered when a column supports one, and **announced** in the run info. **(3) A predictive leak detector**: V6 caught columns that MAP to the target; this catches the merely predictive one — a one-column stump fitted on train and scored on validation, and a lone column reading the target at ≥ 99% shows as a copper warning with its measured score, never as a victory. **(4) A robust leaderboard on demand**: 5×2 repeated cross-validation over train+validation (the pipeline refitted inside every fold, the test set never touched), reporting a mean, a spread, and how often the leader actually beat the runner-up — « 10 of 10 folds: the order is stable » or « 6 of 10: treat them as tied ». **A defect the wave exposed and named**: with the smaller train split, Gaussian Naive Bayes on ~150 TF-IDF features saturates to exactly 0/1, so V24's word-effect occlusion measured exactly zero for every word and the card simply vanished — reading as « no word matters », which is false. Measured (2 distinct probabilities out of 48 test rows, against 48 for logistic and gbdt), the card now **refuses by name** and points at a model that can answer. 369 unit tests, 65 e2e. | Owner request (22/08/2026), launched 23/08/2026. Two of the four items were defects rather than gaps: a lab that sells honest evaluation cannot ship a headline figure it knows to be optimistic, nor a split that leaks on dated data. | @@ -451,6 +451,7 @@ data budget asks. | **V38 — delivered** | **Data Studio: reading the file exactly as it was written.** The headline item was a **defect in shipped code, not a missing feature**, and the wave opened by proving it. Same 900 rows, same three numeric columns, same seed — only the way a number is spelled changes — the figures are in **« V38 — the same file, two spellings »** below this table. The baseline scores 0.594, so the headroom above it collapses from 0.406 to 0.225: **45% of the achievable gain, lost in silence**. `parseNumber` ends in `Number(cleaned)`, `Number('12,5')` is `NaN`, the column falls through, and V24's TF-IDF cheerfully tokenises digits into a hundred word features. Nothing warned, nothing refused — the pipeline just produced a worse model. **Three of this row's own predictions were wrong, and the measurement corrected them**: the mis-typed column becomes `text` on high cardinality but `categorical` (one-hot) on low, not `text` alone; a windows-1252 file does not display `Québec` — that is the reverse case (UTF-8 read as cp1252) — it yields U+FFFD replacement characters; and day-first dates were already handled by V8, `31/12/2025` parsing correctly all along, with only the dash form `31-12-2025` returning null. **The fix.** One reader, shared by the ML Lab and the Data Studio, that decides by evidence rather than by guessing the user's locale: the browser's language says nothing about the file someone dragged in. Encoding is settled by **trying and failing** — UTF-8 in `fatal` mode throws on cp1252 accents, so the fallback is a certainty, not a preference; the delimiter is the candidate that splits every sampled line into the same number of columns, quotes respected; and the decimal separator is decided **per column**, never per file. A column is rewritten only when at least 90% of its values are numbers in that form AND it carries a comma AND it would otherwise not be numeric at all — so a text column containing « vis, tête plate » comes out untouched, and a column of bare integers is left alone because rewriting it would be a change with no cause. Values that do not match the pattern are never rewritten: a stray « n/d » stays « n/d » rather than becoming a plausible-looking number. Detection reads only the **head** of the file, so V25's streaming abort survives intact and a 2 GB file still stops at the cell budget. The reading is then **announced with its evidence** — « virgule décimale détectée et convertie dans 2 colonnes : surface (400/400), prix (400/400) » — above a five-row preview, and only when the reading was not the plain default: an ordinary UTF-8 comma file gets no card, no confirmation step, no friction. Batch scoring, drift and join files go through the same reader, because comparing a French export against a normalised dataset would otherwise report drift that is nothing but a decimal separator. **What this wave deliberately does not do**: guess a locale from `navigator.language` (the file has no relationship to the browser's language), rewrite a column on a bare majority (below the 90% floor the evidence is not evidence), or touch values individually inside a column the evidence does not cover. 446 unit tests, 73 e2e — of the 21 new reader tests, five assert that it **refuses** to act. | Owner request (22/08/2026): what to improve in /data. The audit found a defect first, and the wave began by reproducing it end to end: a French-locale CSV — the single most likely file this owner's users will open — silently loses every numeric column. The repair is exact rather than approximate: the French file now trains to the same types, the same feature count and the same score, to ten decimal places, as the file that never had the problem. | | **V39 — delivered** | **Data Studio: a recipe that works column by column.** `RecipeOptions` applied `missing` and `clipOutliers` to the **whole file** — one strategy for every column, however different they are. A median makes sense for an age and none at all for a postcode. The recipe is now an ordered list of **per-column steps**, with the file-wide settings demoted to **defaults a column may override**: a column with no entry behaves exactly as it did before V39, which is what keeps every previously exported recipe valid and replayable. The missing strategies that were absent are there: **median, mean, most-frequent, a constant used verbatim, and a « MANQUANT » category** for categorical columns — absence becoming a level of its own rather than a guess. Clipping became a per-column decision on the same terms, so a column can opt in or out of the global flag. **The rule the wave exists to enforce: imputing without marking destroys information.** A blank field is rarely blank at random, and the fact of the blank is frequently predictive on its own, so every column may add a `_absent` indicator — and the ordering is the whole point: **every indicator is written before any blank is filled**, so it records where the blanks really were, never where some other column's rule left them. Three consequences fall out of that and are tested: a column with nothing missing gets no indicator (an all-zero column is noise, not information); rows dropped by one column's rule are gathered and removed **once**, so indicators stay aligned instead of being shifted by a sibling column; and when a strategy cannot be honoured — a median over a column holding no parseable number, a constant with no value typed — the blanks are **left blank** rather than filled with something invented. Because the indicator is optional, the studio **announces the columns it filled without marking**, by name, instead of quietly producing a tidier table. **What this wave deliberately does not do**: make the indicator mandatory (it would silently add columns to every existing recipe, and imposing is not the same as announcing), impute from a model (opaque, and it fabricates values that look plausible — already refused for V40), or reorder the recipe's fixed stages, which is what keeps toggling an option from compounding with the last one. 461 unit tests, 75 e2e. | A single global strategy is the kind of default that looks tidy and quietly makes the data worse; per-column steps cost little to build because the recipe was already an object, not a pile of checkboxes. The build confirmed it: the engine change is contained in one stage of `applyRecipe`, and the 17 pre-existing recipe tests passed untouched. | | **V40 — delivered** | **Data Studio: validity, drift, and an auditable diff.** Quality was measured as completeness and consistency of type; what was missing is **validity** — a value can be present, correctly typed and still impossible. Five named rules now say so in plain language: an age outside 0–120, a date in the future, a percentage outside 0–100, a negative amount, a malformed postcode. Then **cross-column consistency**: an end date before its start, a total that is not quantity × price — every cell fine, the ROW impossible. **The plan's choice of engine was wrong, and building it showed why.** It proposed running the consistency rules through V29's DuckDB, since the file is already registered there. But DuckDB is an announced, opt-in 18–22 MB download: routing these checks through it would have made a universally applicable check conditional on a large download most users will decline, leaving the panel empty for them. Comparing two columns of a table already in memory is a loop, so it is a loop, and every user gets it. DuckDB keeps the job it is genuinely needed for — arbitrary SQL, and the new **Parquet export** (`COPY … TO`, one call, bytes straight to a download). Both families of rules obey the two laws V38's reader established: **they fire on evidence, never on a column's name alone** — a column called `age` holding 20 000 is a duration in days, so the rule checks that most of the column is plausible before it flags the rest, and stays silent otherwise — and **they report without ever repairing**, because V39's recipe is the one record of what was done to the data. Three things then make the studio auditable rather than merely helpful. A **before/after diff** naming which rows, which columns and which values changed: the hard part is that a recipe drops rows and adds columns, so `applyRecipe` now returns which SOURCE row each surviving row came from — without it the diff would pair row 7 with a different row 7 and report a screen full of changes that never happened. A **replayable reference profile**, the V22 manifest idea applied to data: bin edges and shares, never rows, so a profile of a payroll file describes the shape of the salary distribution and nobody's salary — which is what makes it safe to commit beside the code, and it scores a new file to the same PSI, to six decimal places, as V11's live two-file comparison. And a **score broken into its parts**, each with its weight and what it actually cost. The weights now sum to **105, not 100**, deliberately: validity brought its own 5 points rather than taking them from an existing part, because redistributing would have quietly changed what every previously published score meant. **What this wave deliberately does not do**: a spreadsheet-style cell editor (hand edits break reproducibility — the recipe is the record), fuzzy deduplication (guaranteed false positives on names and addresses, silently merging two real people), or model-based imputation (opaque, and it fabricates values that look plausible). 502 unit tests, 78 e2e. | Came last because it builds on V38's faithful read and V39's per-column recipe: validity rules on mis-parsed numbers would have flagged the parser, not the data. Closes the Data Studio group. Of the 36 new unit tests, eleven assert that a rule REFUSES to fire — a rule that flags a good file is worse than no rule, because it teaches the reader to ignore the panel. | +| **V41 — delivered** | **The site says what it is.** Six corrections to what the site declared about itself — to crawlers, to link previews, to visitors — none of them touching the lab, all measured on production first (29/09/2026, `curl`). **(1) The twelve `/docs/` pages the sitemap advertised were served by the root fallback**: the home page's title and Open Graph tags, a canonical pointing at `/`, an empty `
`. To a crawler, twelve duplicates of the home page; to a reader pasting a tutorial link in a chat, a preview of the wrong page. V35's guard checked the section shells and the sitemap's status codes and missed it, because a fallback answers 200 too. Each documentation page is now prerendered as `docs/.html` — Pages serves that at the clean URL the sitemap and the app's links already use, where a directory would have added a redirect — carrying the compiled article, its title, its summary and its own canonical; `shells.spec.ts` now reads every URL the sitemap lists. The twelve shells stay out of the service worker's precache (~130 KB each of inlined stylesheet and fonts, for pages the app already renders offline from its bundled `DOCS` module); the extglob `docs/!(index).html` spares the section index, which the plain `docs/*.html` had dropped too — 96 → 95 precache entries, measured. **(2) The home description opened mid-sentence** — « entirely in your browser. Drop a dataset… » — because only the highlighted half of the title was prepended to the lede. **(3) `/` was the one section painting nothing before JavaScript**, and the one most visitors land on, because the root file was also the SPA fallback (V14: « the root stays the SPA fallback (accepted) »). That acceptance is reversed: the root carries the home hero with `HomePage`'s exact classes, and the fallback role moves to two new files. **(4) A real 404.** `wrangler pages dev` reports the file's single rule, `/* /index.html 200`, as **invalid and ignored** — Pages strips `/index.html` from URLs, so the rewrite would loop — and the 200 every unknown address answered came from the platform's default single-page fallback: every misspelled URL was a soft 404, and so was every missing asset, answered with HTML. A `404.html` (the not-found hero, `noindex`) turns that fallback off; `_redirects` lists only the four routes with no file of their own — a run, a comparison, a share link — rewritten to the bare `shell.html` at its clean URL `/shell` (`/shell.html` itself answers a 308), exact source first, splats after, and the service worker's navigation fallback moves to the same shell so an offline run page no longer flashes the home hero. A unit test derives that list from `router.tsx` and the shell routes, so a route added on one side fails on the other; and a new Playwright project runs `routing.spec.ts` against `wrangler pages dev`, the one server that applies `_redirects`, `_headers` and `404.html` the way production does — `vite preview` answers every unknown path with `index.html` and a 200, which is why nothing in the suite could tell a real 404 from a soft one. **(5) The home card said « the three modules are live »** — written at V22, untouched through eighteen waves. It is now « What's new »: a current map of the five areas, the latest wave read from the build version, a link to the log. **(6) A `CHANGELOG.md` extracted from this file** — one entry per wave from V7 on, the 27.x honesty fixes included and V12 (pending) excluded, newest first — by `npm run changelog`, which also aligns `package.json` on `1..0`; a test compares the committed file with the generator's output byte for byte, so the changelog cannot drift from the plan in either direction, and pins the version to the last wave recorded here. Three rows of this file (V25–V27) turned out to have no « why » cell; they are kept with an empty rationale rather than dropped. The V32 row held an unescaped `\|` inside inline code, which split it into four cells and made its published rationale a fragment of its content: the pipe is now escaped, and the generator refuses any wave row with more than three cells — or with a status it does not know — rather than skipping it, because a row skipped in silence leaves the version a wave behind with every test green. Also: the « PS: » about dataset size leaves the`/ml` lede for a note under the drop zone, where the file arrives, and the README's duplicated LIMITATION line is one line. **What this wave deliberately does not do**: translate the changelog (the wave titles are this file's, in English — the card shows the number), prerender the documentation in French (the shells are English like every section; the app switches on mount), or add hreflang alternates (the language is the app's, not the URL's). 817 unit tests, 122 e2e. | Owner request (29/09/2026): an analysis of the repository and of the production site, then a first wave. All six items are defects in what the site said about itself rather than missing features — a lab that publishes its refusals and its limits cannot have its documentation indexed as twelve copies of its home page, nor answer 200 to an address that does not exist. | **V30's measurements**, all on the same 55 questions, the same pinned q4f16 weights and the same CPU runtime. « app » is what a visitor actually gets: the diff --git a/README.md b/README.md index bca581a..c9c94ab 100644 --- a/README.md +++ b/README.md @@ -90,8 +90,6 @@ contributions](docs/screenshots/insights.png) _Every figure above was produced by the app itself, on the `titanic.csv` sample, seed 42 — reproducible by pressing train._ -**LIMITATION: The ideal dataset size is between 1MB and 30MB; beyond 30 MB, the browser response time may take longer to return the results.** - ### Data Studio — `/data` - **Quality report** with a deterministic 0–100 score: missing cells, duplicates, @@ -234,14 +232,18 @@ Wikimedia Commons), the portrait is NASA, public domain._ so), slow model families train on measured, announced caps scored against the same full test set, and parsing refuses past a named 20M-cell memory budget instead of letting the tab die. -- **Performance.** Every section serves a prerendered static shell (hero paints before - JavaScript); Lighthouse mobile ≈ 0.99 on `/ml` under real throttling. Heavy - dependencies (Dexie, SheetJS, ONNX Runtime) load lazily. -- **Quality bar.** 711 unit tests and 111 Playwright end-to-end tests across three browser - projects — desktop, a phone viewport, and dark mode — covering offline PWA, a fake - webcam, a horizontal-overflow guard on every route, and axe-core WCAG A/AA checks on - every page including the twenty-four documentation pages. Plus strict TypeScript, - ESLint, Prettier, and Lighthouse budgets — all enforced in CI. +- **Performance.** Every page — the sections, the home page and each of the twelve + documentation URLs — serves a prerendered static shell (hero paints before JavaScript; + a doc page carries its whole article, in English until the app mounts); Lighthouse + mobile ≈ 0.99 on `/ml` under real throttling. Heavy dependencies (Dexie, SheetJS, ONNX + Runtime) load lazily. +- **Quality bar.** 817 unit tests and 122 Playwright end-to-end tests across five + projects — desktop, a phone viewport in English and in French, dark mode, and + Cloudflare Pages' own routing emulated by `wrangler pages dev` (a real 404, the security + headers as served) — covering offline PWA, a fake webcam, a horizontal-overflow guard on + every route, and axe-core WCAG A/AA checks on every page including the twenty-four + documentation pages. Plus strict TypeScript, ESLint, Prettier, and Lighthouse budgets — + all enforced in CI. - **One dependency does not come from npm.** SheetJS left the registry, and the copy still published there (`xlsx@0.18.5`) carries two unfixable high advisories. The dependency points at the project's official tarball instead, which fixes both; @@ -265,15 +267,16 @@ npm ci # install dependencies npm run dev # start the dev server ``` -| Script | Purpose | -| --------------------------------------- | ---------------------------------- | -| `npm run test` | Unit tests (Vitest) | -| `npm run e2e` | End-to-end tests (Playwright) | -| `npm run typecheck` | TypeScript, strict mode | -| `npm run lint` / `npm run format:check` | ESLint / Prettier | -| `npm run build` | Production build to `dist/` | -| `npm run preview` | Serve the production build locally | -| `npm run llm:prepare` | Fetch and split the local LLM | +| Script | Purpose | +| --------------------------------------- | -------------------------------------------------------------- | +| `npm run test` | Unit tests (Vitest) | +| `npm run e2e` | End-to-end tests (Playwright) | +| `npm run typecheck` | TypeScript, strict mode | +| `npm run lint` / `npm run format:check` | ESLint / Prettier | +| `npm run build` | Production build to `dist/` | +| `npm run preview` | Serve the production build locally | +| `npm run llm:prepare` | Fetch and split the local LLM | +| `npm run changelog` | Regenerate `CHANGELOG.md` from `PLAN.md` and align the version | The language model behind the data assistant is **not committed** (355 MB). `npm run llm:prepare` downloads it into `public/llm/` and splits it into parts under Cloudflare's @@ -321,7 +324,8 @@ CI builds, tests and deploys on every push: pull requests get a Cloudflare Pages Development proceeds in planned "caps" of feature waves; six caps have shipped (MVP through the lab meeting the real world — real photos, real text, real file sizes). The full plan, delivery log and design decisions live in -[PLAN.md](PLAN.md). +[PLAN.md](PLAN.md); [CHANGELOG.md](CHANGELOG.md) is extracted from it — one entry per +wave, newest first — by `npm run changelog`, and a test fails when the two disagree. ## License diff --git a/e2e/routing.spec.ts b/e2e/routing.spec.ts new file mode 100644 index 0000000..5e4ecce --- /dev/null +++ b/e2e/routing.spec.ts @@ -0,0 +1,80 @@ +import { expect, test } from '@playwright/test'; + +/** + * V41 — what Cloudflare Pages does with a URL, checked against Pages' own + * routing rather than the dev server's. + * + * `vite preview` answers every unknown path with `index.html` and a 200, and + * ignores `_redirects`, `_headers` and `404.html` — so nothing in the rest of + * the suite could tell a real 404 from a soft one. This spec runs in the + * `pages` project only, against `wrangler pages dev dist`, which applies the + * same asset routing production does. + * + * Measured on production before V41: every misspelled address answered 200, + * and the file's one rule (`/* /index.html 200`) was silently INVALID — Pages + * strips `/index.html` from URLs, so the rewrite would loop and the parser + * dropped it. The 200 came from the default single-page fallback instead. + */ +test.use({ locale: 'en-US' }); + +test('an unknown address answers 404 with the not-found page, not indexed', async ({ request }) => { + const response = await request.get('/this-address-does-not-exist'); + expect(response.status()).toBe(404); + const html = await response.text(); + expect(html).toContain('Page not found · LabML'); + expect(html).toContain(''); + // A 404 has no canonical: there is nothing there to be the canonical of. + expect(html).not.toContain('rel="canonical"'); +}); + +test('a misspelled documentation slug answers 404 too', async ({ request }) => { + expect((await request.get('/docs/this-page-does-not-exist')).status()).toBe(404); +}); + +test('a missing asset answers 404, never HTML with a 200', async ({ request }) => { + const response = await request.get('/assets/this-chunk-does-not-exist.js'); + expect(response.status()).toBe(404); +}); + +test('a run, a comparison and a share link answer 200 with the bare shell', async ({ request }) => { + for (const path of ['/ml/run/abc', '/ml/compare/a/b', '/ml/compare-many/a,b,c', '/ml/share']) { + const response = await request.get(path); + expect(response.status(), path).toBe(200); + const html = await response.text(); + // No hero: these pages describe one visitor's local data and the app + // paints them; a hero would show the wrong content for a frame. + expect(html, path).not.toMatch(/]/); + expect(html, path).toContain(''); + // Previews still work — a share link pasted in a chat shows the site card. + expect(html, path).toContain('
'); + } +}); + +test('the home page, a section and a documentation page answer 200 with their hero', async ({ + request, +}) => { + const pages: [string, string][] = [ + ['/', 'A machine learning lab,'], + ['/ml/', 'From a CSV to a leaderboard'], + [ + '/docs/premier-modele', + 'rel="canonical" href="https://app.dominicdapice.com/docs/premier-modele"', + ], + ]; + for (const [path, text] of pages) { + const response = await request.get(path); + expect(response.status(), path).toBe(200); + const html = await response.text(); + expect(html, path).toContain(text); + expect(html, path).toMatch(/]/); + } +}); + +test('the security headers are served by the asset routing, not only declared', async ({ + request, +}) => { + const headers = (await request.get('/')).headers(); + expect(headers['content-security-policy']).toContain("default-src 'self'"); + expect(headers['x-frame-options']).toBe('DENY'); +}); diff --git a/e2e/shells.spec.ts b/e2e/shells.spec.ts index db3a01e..e3f04a8 100644 --- a/e2e/shells.spec.ts +++ b/e2e/shells.spec.ts @@ -81,6 +81,58 @@ test('every shell carries its own title, description and social card', async ({ expect(og.headers()['content-type']).toContain('image/png'); }); +/** + * V41 — every URL the sitemap advertises must describe itself. + * + * Measured on production (29 Sep 2026): the twelve `/docs/` pages the + * sitemap listed were all served by the root fallback — the home page's title, + * description, Open Graph tags and canonical, and an empty `
`. + * To a crawler that is twelve duplicates of the home page; to a reader pasting + * a tutorial link into a chat, a preview of the wrong page. The V35 guard + * above checked the section shells and the sitemap's status codes, and missed + * it because a fallback answers 200 too. This one reads every advertised URL. + */ +test('every page the sitemap advertises describes itself without JavaScript', async ({ + request, +}) => { + const xml = await (await request.get('/sitemap.xml')).text(); + const listed = [...xml.matchAll(/https:\/\/app\.dominicdapice\.com([^<]*)<\/loc>/g)].map( + (match) => match[1], + ); + expect(listed.length).toBeGreaterThanOrEqual(21); + + const titles = new Map(); + for (const path of listed) { + const html = await (await request.get(path)).text(); + const meta = (pattern: RegExp) => pattern.exec(html)?.[1]?.trim() ?? ''; + + expect(meta(/([^<]*)<\/title>/); + expect(titles.has(title), `${path} repeats the title of ${titles.get(title)}`).toBe(false); + titles.set(title, path); + expect( + meta(/]/); + } +}); + +test('the home page description starts where its sentence starts', async ({ request }) => { + const html = await (await request.get('/')).text(); + const description = / { const xml = await (await request.get('/sitemap.xml')).text(); const listed = [...xml.matchAll(/https:\/\/app\.dominicdapice\.com([^<]*)<\/loc>/g)].map( @@ -124,9 +176,11 @@ test('the sitemap lists every page that exists, and nothing that does not', asyn * index. There is no error and no visible symptom, which is exactly why this * belongs in a test rather than in someone's memory. * - * The check is on the CONTENT, never the status code: `_redirects` sends every - * unknown path to `index.html` with HTTP 200, so a missing file still answers - * 200 — with HTML. That trap cost a false positive during the V35 audit. + * The check is on the CONTENT, never the status code: `vite preview`, which + * serves this suite, answers every unknown path with `index.html` and a 200, + * so a missing file would still answer 200 — with HTML. That trap cost a false + * positive during the V35 audit. (Production answers a real 404 since V41; + * `routing.spec.ts` checks that side against Pages' own routing.) */ test('the Bing site verification file is served from the root', async ({ request }) => { const response = await request.get('/BingSiteAuth.xml'); diff --git a/package-lock.json b/package-lock.json index d423aaf..78064f6 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "labml", - "version": "0.1.0", + "version": "1.41.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "labml", - "version": "0.1.0", + "version": "1.41.0", "license": "MIT", "dependencies": { "@duckdb/duckdb-wasm": "1.28.0", diff --git a/package.json b/package.json index 0a27a95..bdc93bc 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "labml", "private": true, - "version": "1.0.0", + "version": "1.41.0", "license": "MIT", "type": "module", "engines": { @@ -15,6 +15,7 @@ "llm:bench": "node scripts/run-llm-bench.mjs", "llm:bench:node": "LABML_LLM_BENCH=1 vitest run src/features/ai/llm/bench.node.test.ts", "llm:fetch": "node scripts/prepare-llm.mjs .llm-cache --flat", + "changelog": "node scripts/changelog.mjs", "lint": "eslint .", "format": "prettier --write .", "format:check": "prettier --check .", diff --git a/playwright.config.ts b/playwright.config.ts index 61f5149..3ba75cc 100644 --- a/playwright.config.ts +++ b/playwright.config.ts @@ -26,7 +26,7 @@ export default defineConfig({ // still runs once, and only the specs that can actually catch a viewport or // a theme regression are replayed. projects: [ - { name: 'chromium', use: { ...devices['Desktop Chrome'] } }, + { name: 'chromium', use: { ...devices['Desktop Chrome'] }, testIgnore: /routing\.spec\.ts/ }, { name: 'mobile', testMatch: /(layout|a11y)\.spec\.ts/, @@ -58,10 +58,31 @@ export default defineConfig({ testMatch: /a11y\.spec\.ts/, use: { ...devices['Desktop Chrome'], colorScheme: 'dark' }, }, + { + // V41 — Cloudflare Pages' routing, emulated. `vite preview` serves + // `index.html` with a 200 for every unknown path and ignores + // `_redirects`, `_headers` and `404.html`, so the suite above cannot + // tell a real 404 from a soft one, nor see the headers production + // sends. `wrangler pages dev` applies the same asset routing Pages + // does; only the spec that needs it runs there. + name: 'pages', + testMatch: /routing\.spec\.ts/, + use: { ...devices['Desktop Chrome'], baseURL: 'http://127.0.0.1:8788' }, + }, + ], + webServer: [ + { + command: 'npm run preview', + url: 'http://127.0.0.1:4173', + reuseExistingServer: !process.env.CI, + }, + { + command: 'npx wrangler pages dev dist --port 8788 --ip 127.0.0.1', + url: 'http://127.0.0.1:8788/', + reuseExistingServer: !process.env.CI, + timeout: 120_000, + // No telemetry from the test runner, and no interactive prompt. + env: { WRANGLER_SEND_METRICS: 'false', CI: '1' }, + }, ], - webServer: { - command: 'npm run preview', - url: 'http://127.0.0.1:4173', - reuseExistingServer: !process.env.CI, - }, }); diff --git a/public/_redirects b/public/_redirects index 7797f7c..aa4772a 100644 --- a/public/_redirects +++ b/public/_redirects @@ -1 +1,23 @@ -/* /index.html 200 +# V41 — Cloudflare Pages applies these rules BEFORE it looks for a file +# (« redirects are always followed, regardless of whether or not an asset +# matches »), then serves the exact file if one exists (the prerendered shells, +# the documentation pages, every asset), then 404.html. So a rule must never +# overlap a real file: `/ml/* /shell 200` would hide /ml/index.html in +# production. Only the routes that have no file of their own are listed: a +# run, a comparison, a share link — pages that describe one visitor's local +# data and that the app paints. They are handed the bare shell at its clean +# URL, with a 200 so the address in the bar stays what the visitor typed. +# +# Before V41 the single rule was `/* /index.html 200`. `wrangler pages dev` +# reports it as INVALID and ignores it — Pages strips `/index.html` from URLs, +# so the rewrite would loop — and the 200 on every unknown address came from +# the default single-page fallback instead: a misspelled URL was a soft 404. +# With a 404.html in the build that fallback is off, so this list must be +# exact; `src/app/redirects.test.ts` keeps it aligned with the router, and +# `e2e/routing.spec.ts` replays it against Pages' own routing. +# +# Exact sources first, splats after — the order the Pages parser asks for. +/ml/share /shell 200 +/ml/run/* /shell 200 +/ml/compare/* /shell 200 +/ml/compare-many/* /shell 200 diff --git a/scripts/changelog.d.mts b/scripts/changelog.d.mts new file mode 100644 index 0000000..5e67522 --- /dev/null +++ b/scripts/changelog.d.mts @@ -0,0 +1,12 @@ +export interface Wave { + version: string; + title: string; + summary: string; + why: string; +} + +export function extractWaves(plan: string): Wave[]; +export function renderChangelog(waves: Wave[]): string; +export function packageVersionFor(waves: Wave[]): string; +export function syncPackageVersion(packageJson: string, version: string): string; +export function syncLockVersion(lock: string, version: string): string; diff --git a/scripts/changelog.mjs b/scripts/changelog.mjs new file mode 100644 index 0000000..e78fac9 --- /dev/null +++ b/scripts/changelog.mjs @@ -0,0 +1,205 @@ +/** + * V41 — the CHANGELOG, extracted from `PLAN.md` rather than written by hand. + * + * PLAN.md is the engineering log: every wave has a row in one of its roadmap + * tables, with the wave's content and the reason it was built. At 195 KB it is + * not something a visitor reads, so this script reads it instead and writes one + * entry per wave, newest first. The rule is the one V34 set for the limits + * page: a record recalled from memory flatters, a record extracted from the + * source cannot. `src/lib/changelog.test.ts` fails when `CHANGELOG.md` and the + * plan disagree, in either direction. + * + * npm run changelog + * + * Also aligns the version in `package.json` (and the lockfile's root entries) + * on `1..0`, which is what the home page reads to say « V40 ». + */ +import { readFileSync, writeFileSync } from 'node:fs'; +import { pathToFileURL } from 'node:url'; + +/** @typedef {{ version: string; title: string; summary: string; why: string }} Wave */ + +const ROW = /^\|(.*)\|\s*$/; +/** `V7`, `V27.1 — delivered`, `V12 — pending` — after the bold markers are gone. */ +const WAVE = /^V(\d+(?:\.\d+)?)(?:\s*—\s*(delivered|pending))?$/; +/** + * A cell that opens like a wave row and is not one the parser reads: a bare + * version, or a version followed by some dash — an en dash or a hyphen where + * the plan writes an em dash, a status other than the two the plan uses. The + * benchmark tables' rows (« V27.3 as shipped — … ») put a word after the + * version and are left alone. + */ +const WAVE_LIKE = /^V\d+(?:\.\d+)?(?:\s*[—–-]|$)/; +/** + * A sentence ends at a period followed by whitespace, optionally closing a + * bold run first (`itself.** Two`) so the bold stays balanced. A period inside + * a decimal (`0.818`), a version (`V27.1`) or a file name (`x.csv`) has no + * whitespace after it and ends nothing. + */ +const SENTENCE_END = /\.(\*\*)?(?=\s)/; + +/** + * @param {string} plan + * @returns {Wave[]} + */ +export function extractWaves(plan) { + /** @type {Wave[]} */ + const waves = []; + for (const line of plan.split(/\r?\n/)) { + const row = ROW.exec(line); + if (!row) continue; + // Split on pipes that are not escaped, then drop the escape: `\|` is how + // Markdown writes a pipe inside a cell. + const cells = row[1].split(/(? cell.trim().replace(/\\\|/g, '|')); + if (cells.length < 2) continue; + const label = cells[0].replace(/\*/g, '').trim(); + const head = WAVE.exec(label); + if (!head) { + // A row that names a wave and cannot be read is refused rather than + // skipped: skipped, it leaves the version a wave behind with every test + // green. Statuses other than the two the plan uses are a question for a + // person, not a guess for a script. + if (WAVE_LIKE.test(label)) throw new Error(`PLAN.md: cannot read wave row « ${label} »`); + continue; + } + if (head[2] === 'pending') continue; + // An unescaped pipe inside a cell splits the row and turns the tail of the + // content into the « why » — measured on the V32 row, whose inline code + // held one. The byte-for-byte test cannot see it, so the generator refuses. + if (cells.length > 3) { + throw new Error( + `PLAN.md: wave V${head[1]} has ${cells.length} cells — an unescaped | inside a cell? Write it as \\|`, + ); + } + const content = cells[1]; + const title = titleOf(content); + // Three rows of the plan (V25–V27) never got a « why » cell; an empty + // rationale is reported as such rather than dropping the wave. + waves.push({ + version: head[1], + title, + summary: summaryOf(content, title), + why: cells[2] ?? '', + }); + } + return waves.sort((a, b) => compareVersions(b.version, a.version)); +} + +/** @param {string} content */ +function titleOf(content) { + const bold = /\*\*(.+?)\*\*/.exec(content); + const raw = bold ? bold[1] : content.split(':')[0]; + return plain(raw); +} + +/** + * The first sentence — unless that sentence is the bold title and nothing + * else, in which case the heading would be repeated and the sentence after it + * is the one that says what the wave did. + * @param {string} content + * @param {string} title + */ +function summaryOf(content, title) { + const first = firstSentence(content); + if (plain(first) !== title) return first; + const rest = content.slice(first.length).trim(); + return rest === '' ? first : firstSentence(rest); +} + +/** @param {string} content */ +function firstSentence(content) { + const end = SENTENCE_END.exec(content); + return end ? content.slice(0, end.index + end[0].length) : content; +} + +/** Markdown bold and trailing punctuation removed, for comparing a title to a sentence. */ +function plain(text) { + return text + .replace(/\*\*/g, '') + .trim() + .replace(/[.:—-]+$/, '') + .trim(); +} + +/** + * `28` sorts after `27.3`, which sorts after `27`. + * @param {string} a + * @param {string} b + */ +function compareVersions(a, b) { + const [aMajor, aMinor = 0] = a.split('.').map(Number); + const [bMajor, bMinor = 0] = b.split('.').map(Number); + return aMajor - bMajor || aMinor - bMinor; +} + +/** + * @param {Wave[]} waves newest first + * @returns {string} + */ +export function renderChangelog(waves) { + const head = [ + '# Changelog', + '', + 'Generated from the roadmap tables in `PLAN.md` by `npm run changelog` — not edited by', + 'hand: a unit test fails when this file and the plan disagree, in either direction. One', + 'entry per wave, newest first; the first six waves (V1–V6, the MVP) predate the tables and', + 'are recorded in PLAN.md §B and §J. Each entry carries the opening sentence of its row and', + 'the reason the wave was built.', + '', + ]; + const body = waves.flatMap((wave) => [ + `## V${wave.version} — ${wave.title}`, + '', + wave.summary, + '', + ...(wave.why === '' ? [] : [`_Why:_ ${wave.why}`, '']), + ]); + return [...head, ...body].join('\n'); +} + +/** + * `1..0` — major 1 because nothing here breaks a + * consumer, minor for the wave, patch always 0. Sub-waves (27.1, 27.2…) are + * honesty fixes to their wave and do not move the number. + * @param {Wave[]} waves + */ +export function packageVersionFor(waves) { + const latest = Math.max(...waves.map((wave) => parseInt(wave.version, 10))); + return `1.${latest}.0`; +} + +/** + * Rewrites the first `"version"` field only, leaving the file otherwise + * byte-identical — `package.json` is hand-formatted and diffs should say + * « version », not « reformatted ». + * @param {string} packageJson + * @param {string} version + */ +export function syncPackageVersion(packageJson, version) { + return packageJson.replace(/("version":\s*")[^"]*(")/, `$1${version}$2`); +} + +/** + * The lockfile names the root package twice (top level and `packages[""]`); + * both carry a version, and only those two — every dependency has its own name. + * @param {string} lock + * @param {string} version + */ +export function syncLockVersion(lock, version) { + return lock.replace(/("name": "labml",\s*"version": ")[^"]*(")/g, `$1${version}$2`); +} + +function main() { + const waves = extractWaves(readFileSync('PLAN.md', 'utf8')); + if (waves.length === 0) throw new Error('PLAN.md holds no wave row'); + writeFileSync('CHANGELOG.md', renderChangelog(waves)); + const version = packageVersionFor(waves); + writeFileSync('package.json', syncPackageVersion(readFileSync('package.json', 'utf8'), version)); + writeFileSync( + 'package-lock.json', + syncLockVersion(readFileSync('package-lock.json', 'utf8'), version), + ); + console.log(`CHANGELOG.md: ${waves.length} waves, latest V${waves[0].version} → ${version}`); +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) main(); diff --git a/src/app/redirects.test.ts b/src/app/redirects.test.ts new file mode 100644 index 0000000..c5bf6c4 --- /dev/null +++ b/src/app/redirects.test.ts @@ -0,0 +1,72 @@ +import { readFileSync } from 'node:fs'; +import { describe, expect, it } from 'vitest'; + +/** + * V41 — `public/_redirects` and `src/app/router.tsx` describe the same routes, + * from two sides. The router says which paths the app answers; the redirects + * file says which of them Cloudflare Pages must hand to the app because no + * static file exists for them (a run, a comparison, a share link). A route + * added to one and not the other is a page that renders in the dev server and + * answers 404 in production — the kind of gap nothing else would report. + * + * Until V41 the file held `/* /index.html 200`, which `wrangler pages dev` + * reports as an INVALID rule (Pages strips `/index.html`, so the rewrite loops + * and is ignored): the 200 on unknown URLs came from the default SPA fallback, + * and every misspelled address was a soft 404. A `404.html` now takes that + * role with the right status, so the rewrite list must be exact. + */ +const redirects = readFileSync('public/_redirects', 'utf8'); +const router = readFileSync('src/app/router.tsx', 'utf8'); +const viteConfig = readFileSync('vite.config.ts', 'utf8'); + +const rules = redirects + .split(/\r?\n/) + .map((line) => line.trim()) + .filter((line) => line !== '' && !line.startsWith('#')) + .map((line) => { + const [source, destination, status] = line.split(/\s+/); + return { source, destination, status }; + }); + +/** Every `path: '…'` the router declares, as written. */ +const routerPaths = [...router.matchAll(/path: '([^']+)'/g)].map((match) => match[1]); +/** Every section with a prerendered shell (`dir: '…'` in vite.config.ts). */ +const shellDirs = [...viteConfig.matchAll(/\bdir: '([^']+)'/g)].map((match) => match[1]); + +/** `ml/run/:id` → `/ml/run/*`, `ml/share` → `/ml/share`. */ +function toSource(path: string): string { + const param = path.indexOf(':'); + return `/${param === -1 ? path : `${path.slice(0, param)}*`}`; +} + +describe('_redirects', () => { + it('has no catch-all: unknown addresses must reach 404.html with a 404', () => { + expect(rules.map((rule) => rule.source)).not.toContain('/*'); + }); + + it('rewrites every app route that has no static file, and nothing else', () => { + const expected = routerPaths + .filter((path) => path !== '/' && path !== '*') + // Sections with a shell and doc pages are exact files on disk. + .filter((path) => !shellDirs.includes(path) && !path.startsWith('docs')) + .map(toSource) + .sort(); + expect(rules.map((rule) => rule.source).sort()).toEqual(expected); + expect(expected.length).toBeGreaterThanOrEqual(4); + }); + + it('serves the bare shell at its clean URL, with a 200, for each of them', () => { + for (const rule of rules) { + // `/shell.html` would be answered with a 308 to `/shell` — the very + // loop that made the old rule invalid. The clean URL is the asset. + expect(rule.destination, rule.source).toBe('/shell'); + expect(rule.status, rule.source).toBe('200'); + } + }); + + it('lists exact sources before splats, as the Pages parser recommends', () => { + const firstSplat = rules.findIndex((rule) => rule.source.includes('*')); + const lastExact = rules.map((rule) => rule.source.includes('*')).lastIndexOf(false); + expect(lastExact).toBeLessThan(firstSplat); + }); +}); diff --git a/src/features/home/HomePage.test.tsx b/src/features/home/HomePage.test.tsx new file mode 100644 index 0000000..143ca4a --- /dev/null +++ b/src/features/home/HomePage.test.tsx @@ -0,0 +1,57 @@ +import { render, screen } from '@testing-library/react'; +import { readFileSync } from 'node:fs'; +import { MemoryRouter } from 'react-router'; +import { beforeEach, describe, expect, it } from 'vitest'; +import { HomePage } from '@/features/home/HomePage'; +import i18n from '@/lib/i18n'; + +function renderHome() { + return render( + + + , + ); +} + +/** + * V41 — the home page's status card. It said « the three modules are live » + * and listed them, a sentence written at V22 and never touched through the + * eighteen waves that followed: SQL, the local chat, the forecasts, the + * documentation, the privacy page. The card now names the latest delivered + * wave from the build's version — which `changelog.test.ts` pins to the last + * wave PLAN.md records — and points at the CHANGELOG for the rest. + */ +describe('HomePage — what is new', () => { + const { version } = JSON.parse(readFileSync('package.json', 'utf8')) as { version: string }; + const wave = `V${version.split('.')[1]}`; + + beforeEach(async () => { + await i18n.changeLanguage('en'); + }); + + it('names the latest delivered wave, read from the build version', () => { + renderHome(); + expect(screen.getByText(new RegExp(`\\b${wave}\\b`))).toBeInTheDocument(); + }); + + it('links to the CHANGELOG on the repository', () => { + renderHome(); + expect(screen.getByRole('link', { name: /changelog/i })).toHaveAttribute( + 'href', + 'https://github.com/dapiced/LabML/blob/main/CHANGELOG.md', + ); + }); + + it('no longer counts three modules', () => { + renderHome(); + expect(screen.queryByText(/three modules/i)).toBeNull(); + expect(screen.getByText("What's new")).toBeInTheDocument(); + }); + + it('says the same in French', async () => { + await i18n.changeLanguage('fr'); + renderHome(); + expect(screen.getByText('Quoi de neuf')).toBeInTheDocument(); + expect(screen.getByText(new RegExp(`\\b${wave}\\b`))).toBeInTheDocument(); + }); +}); diff --git a/src/features/home/HomePage.tsx b/src/features/home/HomePage.tsx index 3fe3d59..2be50d3 100644 --- a/src/features/home/HomePage.tsx +++ b/src/features/home/HomePage.tsx @@ -4,6 +4,15 @@ import { Link } from 'react-router'; import { Card } from '@/components/ui/card'; import { Eyebrow } from '@/components/ui/eyebrow'; +/** + * V41 — `1..0`: the minor is the latest delivered wave, kept there by + * `npm run changelog` and pinned to PLAN.md by its test. The card used to say + * « the three modules are live », a sentence written at V22 and untouched + * through the eighteen waves that followed. + */ +const LATEST_WAVE = `V${__APP_VERSION__.split('.')[1]}`; +const CHANGELOG_URL = 'https://github.com/dapiced/LabML/blob/main/CHANGELOG.md'; + export function HomePage() { const { t } = useTranslation(); @@ -54,6 +63,16 @@ export function HomePage() { {t('home.statusTitle')}

{t('home.statusBody')}

+

+ {t('home.latestWave')} {LATEST_WAVE} + {' · '} + + {t('home.changelog')} + +

diff --git a/src/features/ml/components/DropZone.tsx b/src/features/ml/components/DropZone.tsx index 61224a7..8b45ac4 100644 --- a/src/features/ml/components/DropZone.tsx +++ b/src/features/ml/components/DropZone.tsx @@ -40,6 +40,8 @@ export function DropZone() {