From 87c1ba9a7184596534a74bc7f3e001d3688d6f07 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 06:09:46 +0000 Subject: [PATCH 01/23] suggest-diff.py (check 18: every machine-applicable suggestion applied and compiled); finding 29: lint fixes that break builds Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- docs/hunt.md | 1 + docs/hunt/lint-fixes-break-builds.md | 50 +++++ .../tests/lint-fixes/apply-one-suggestion.py | 16 ++ docs/hunt/tests/lint-fixes/closure.rs | 2 + docs/hunt/tests/lint-fixes/glob.rs | 7 + docs/hunt/tests/lint-fixes/ormut.rs | 6 + docs/hunt/tests/lint-fixes/ref.rs | 6 + docs/hunt/tests/lint-fixes/unreach.rs | 3 + docs/hunt/tests/lint-fixes/update.rs | 3 + rustc/suggest-diff.py | 195 ++++++++++++++++++ 10 files changed, 289 insertions(+) create mode 100644 docs/hunt/lint-fixes-break-builds.md create mode 100644 docs/hunt/tests/lint-fixes/apply-one-suggestion.py create mode 100644 docs/hunt/tests/lint-fixes/closure.rs create mode 100644 docs/hunt/tests/lint-fixes/glob.rs create mode 100644 docs/hunt/tests/lint-fixes/ormut.rs create mode 100644 docs/hunt/tests/lint-fixes/ref.rs create mode 100644 docs/hunt/tests/lint-fixes/unreach.rs create mode 100644 docs/hunt/tests/lint-fixes/update.rs create mode 100644 rustc/suggest-diff.py diff --git a/docs/hunt.md b/docs/hunt.md index 69e35b7..624a06e 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -52,6 +52,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 26 | meilisearch (edition 2021) stops compiling on nightly-2026-10-06: `Ok(()) as Result<_>` now infers `!` (never-type fallback in edition 2021), and neither 1.98 nor the July nightly warned, also with the future-compatibility lints on | **looks new** as a lint false negative; found by release-to-release; reduced to 11 lines; [facts](hunt/release-regressions.md) | | 27 | under the new trait solver (nightly's default), a type parameter that appears only in a projection (`T0::Of<'_>`) of a function-pointer coercion is not inferred (E0283); breaks surrealdb through `diskann-wide 0.54.0` | **known, intended**: `diskann-wide` is listed in #160895 ("higher-ranked associated type", the intended breakage of trait-system-refactor-initiative#168; 0.55 not yet patched); surrealdb is an affected project not on that list; found by release-to-release; [facts](hunt/release-regressions.md) | | 28 | glob-import ambiguity depends on item order: with two modules re-exporting each other's globs, one order is E0659 and the other compiles and calls a different function (1.98, nightly); the accepted order has swapped between releases | **looks new**; stable code; found by the equivalent-rewrite differential (`reorder`) on 3 UI tests; [facts](hunt/glob-ambiguity-order.md) | +| 29 | machine-applicable lint fixes (what `cargo fix` applies unasked) break builds: `unused_variables` turns `ref b` into a moving `_b` and renames only the declaration of variables mentioned elsewhere, `unused_mut` changes one or-pattern alternative or a variable a `move` closure assigns, `unused_imports` removes a glob that resolution needs | **looks new**; stable 1.98; found by the suggestions-apply check (111 lint fixes in UI tests); six 3–7 line reductions; [facts](hunt/lint-fixes-break-builds.md) | Findings 1 and 2 are single-threaded: an ordinary `cargo build`, an edit, another `cargo build`, and the metadata differs from a clean build of the edited source. Both come diff --git a/docs/hunt/lint-fixes-break-builds.md b/docs/hunt/lint-fixes-break-builds.md new file mode 100644 index 0000000..17230aa --- /dev/null +++ b/docs/hunt/lint-fixes-break-builds.md @@ -0,0 +1,50 @@ +# Machine-applicable lint fixes that break builds + +Facts for finding 29. Found by the suggestions-apply check (`rustc/suggest-diff.py`, check 18 in +[`checks.md`](../checks.md)): every `MachineApplicable` suggestion of every UI test without +`//@ run-rustfix` (17,945 tests, 7,385 suggestions), each applied alone and compiled again. A +lint's machine-applicable fix is what `cargo fix` and `cargo clippy --fix` apply without asking; +a lint never stops a build, so its fix must not introduce an error. 111 lint fixes did. + +## Six shapes, reduced, on stable 1.98.0 + +Each file in [`tests/lint-fixes/`](tests/lint-fixes/) compiles with warnings; applying the one +named suggestion (`apply-one-suggestion.py 1.98.0`) gives the error shown. The same on +nightly-2026-10-06. + +| file | lint | code | suggestion | after applying it | +|---|---|---|---|---| +| `ref.rs` | `unused_variables` | `let ref b = u; drop(u);` | `ref b` → `_b` | E0382 use of moved value: the binding now moves | +| `update.rs` | `unused_variables` | `fn test(f: Foo) { Foo { foo: 4, ..f } }` (every field given) | `f` → `_f` | E0425 cannot find value `f`: the base still names it | +| `unreach.rs` | `unused_variables` | `let x = f(); let _ = x;` with `f() -> Never` | `x` → `_x` | E0425: the later (unreachable) use still names `x` | +| `closure.rs` | `unused_variables` | `let mut x = 0; to_fn(move \|\| { x = 42; })` | `mut x` → `_x` | E0425: the closure still names `x` | +| `closure.rs` | `unused_mut` | the same | remove `mut` | E0594 cannot assign to `x`: the closure assigns | +| `ormut.rs` | `unused_mut` | `Ok(mut y) \| &Err(mut y) => drop(y)` | remove one `mut` | E0409 bound inconsistently across alternatives | +| `glob.rs` | `unused_imports` | `mod one_private { use crate::m::*; pub use crate::m::*; } use crate::one_private::S;` | remove `pub use crate::m::*;` | E0603 struct import `S` is private | + +In the UI tests the same shapes appear as: `ref`/`ref mut` bindings, including `ref x @ pat` and +unsized `ref rest @ ..`, where E0277 follows (25 suggestions); variables mentioned again only in +unreachable code, struct-update bases or closures (48); or-patterns where one alternative's +`mut` or name is changed alone (12); glob re-exports that take part in ambiguity or visibility +(5). + +## Expected + +A machine-applicable suggestion keeps the program compiling, with the same meaning: +`ref _b` (or `_`) for a `ref` binding, renaming every mention or not offering the rename where +the variable is mentioned elsewhere, removing `mut` from every alternative, and not calling an +import unused when removing it changes resolution. + +## Also found, lower priority + +- Lint suggestions inside macro input that the macro then fails to match: `unexpected_cfgs` + (`FALSE` → `false` in a macro's meta argument), `missing_abi` next to a literal with a suffix, + `unused_parens` around `let` chains passed to a macro. +- 106 error suggestions that leave the same error, and 63 whose fix no longer parses, mostly in + parser-recovery and `fn_delegation` tests (error-recovery suggestions on deliberately broken + code). +- An ICE after applying an E0308 fix (`consts/const-eval/array-len-mismatch-type.rs`, stable + 1.98): const evaluation runs on a body that failed with E0277 and panics with "expected wide + pointer extra data"; that ICE family has open #154779. + +Not found in the issue tracker (searched for each shape). diff --git a/docs/hunt/tests/lint-fixes/apply-one-suggestion.py b/docs/hunt/tests/lint-fixes/apply-one-suggestion.py new file mode 100644 index 0000000..1f8fdea --- /dev/null +++ b/docs/hunt/tests/lint-fixes/apply-one-suggestion.py @@ -0,0 +1,16 @@ +import json,subprocess,sys +f=sys.argv[1]; tc=sys.argv[2] +r=subprocess.run(['rustc','+'+tc,'--edition','2021','--emit=metadata','--error-format=json','-o','/dev/null',f],capture_output=True,text=True) +src=open(f,'rb').read(); n=0 +for l in r.stderr.splitlines(): + try: d=json.loads(l) + except: continue + for c in [d]+d.get('children',[]): + parts=[(s['byte_start'],s['byte_end'],s['suggested_replacement']) for s in c.get('spans',[]) if s.get('suggestion_applicability')=='MachineApplicable' and s.get('suggested_replacement') is not None] + if not parts or d['level']!='warning': continue + n+=1; fixed=bytearray(src) + for a,b,t in sorted(parts,reverse=True): fixed[a:b]=t.encode() + g=f.replace('.rs',f'_fix{n}.rs'); open(g,'wb').write(fixed) + rr=subprocess.run(['rustc','+'+tc,'--edition','2021','--emit=metadata','-o','/dev/null',g],capture_output=True,text=True) + errs=[x for x in rr.stderr.splitlines() if x.startswith('error')] + print(f" {d['code']['code'] if d.get('code') else '-'}: {[src[a:b].decode() for a,b,_ in parts]} -> {[t for _,_,t in parts]} => {errs[0] if errs else 'compiles'}") diff --git a/docs/hunt/tests/lint-fixes/closure.rs b/docs/hunt/tests/lint-fixes/closure.rs new file mode 100644 index 0000000..7ad537c --- /dev/null +++ b/docs/hunt/tests/lint-fixes/closure.rs @@ -0,0 +1,2 @@ +fn to_fn(f: F) -> F { f } +fn main() { let mut x = 0; let _f = to_fn(move || { x = 42; }); } diff --git a/docs/hunt/tests/lint-fixes/glob.rs b/docs/hunt/tests/lint-fixes/glob.rs new file mode 100644 index 0000000..435d2c2 --- /dev/null +++ b/docs/hunt/tests/lint-fixes/glob.rs @@ -0,0 +1,7 @@ +mod m { pub struct S {} } +mod one_private { + use crate::m::*; + pub use crate::m::*; +} +use crate::one_private::S; +fn main() { let _ = S {}; } diff --git a/docs/hunt/tests/lint-fixes/ormut.rs b/docs/hunt/tests/lint-fixes/ormut.rs new file mode 100644 index 0000000..b3af771 --- /dev/null +++ b/docs/hunt/tests/lint-fixes/ormut.rs @@ -0,0 +1,6 @@ +fn main() { + let res: &Result = &Ok(1); + match res { Ok(mut x) | &Err(mut x) => { x += 1; drop::(x) } } + let r2: &Result = &Ok(1); + match r2 { Ok(mut y) | &Err(mut y) => drop::(y) } +} diff --git a/docs/hunt/tests/lint-fixes/ref.rs b/docs/hunt/tests/lint-fixes/ref.rs new file mode 100644 index 0000000..8e7780e --- /dev/null +++ b/docs/hunt/tests/lint-fixes/ref.rs @@ -0,0 +1,6 @@ +struct U; +fn main() { + let u = U; + let ref b = u; // unused `b`: suggestion replaces `ref b` with `_b`, which moves `u` + drop(u); +} diff --git a/docs/hunt/tests/lint-fixes/unreach.rs b/docs/hunt/tests/lint-fixes/unreach.rs new file mode 100644 index 0000000..2818afb --- /dev/null +++ b/docs/hunt/tests/lint-fixes/unreach.rs @@ -0,0 +1,3 @@ +enum Never {} +fn f() -> Never { panic!() } +fn main() { let x = f(); let _ = x; } diff --git a/docs/hunt/tests/lint-fixes/update.rs b/docs/hunt/tests/lint-fixes/update.rs new file mode 100644 index 0000000..4c6b87d --- /dev/null +++ b/docs/hunt/tests/lint-fixes/update.rs @@ -0,0 +1,3 @@ +struct Foo { foo: i32 } +fn test(f: Foo) -> i32 { let g = Foo { foo: 4, ..f }; g.foo } // `f` reported unused +fn main() { test(Foo { foo: 1 }); } diff --git a/rustc/suggest-diff.py b/rustc/suggest-diff.py new file mode 100644 index 0000000..1112b61 --- /dev/null +++ b/rustc/suggest-diff.py @@ -0,0 +1,195 @@ +#!/usr/bin/env python3 +"""Suggestions apply: a machine-applicable suggestion must produce code that compiles the way the +suggestion promises. + +For each standalone UI test, collects the diagnostics rustc emits (`--error-format=json`) and, for +each one carrying a `MachineApplicable` suggestion inside the test file, applies that one +suggestion (all its parts) to a copy and compiles the copy again. Findings: + + lint-breaks the suggestion came from a warning (a lint) and the fixed file has an error the + original did not: a lint's fix must never break a build + parse the fixed file no longer parses + not-fixed the same diagnostic (code and message) is reported again at the edited place + ice the fixed file crashes the compiler + +Errors appearing after an *error's* suggestion is applied are expected (compilation gets further) +and are not reported. Tests with `//@ run-rustfix` are skipped by default: compiletest already +checks their fixes. + + rustc/suggest-diff.py --rustc --tests /tests/ui --work [--with-rustfix] + [--max 8] [--only ] [--known ] [--jobs 8] [--pause-on-finding] [--recheck] +""" + +import argparse +import json +import os +import re +import shutil +import subprocess +import sys +import tempfile +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parent)) +import uitest # noqa: E402 + +KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) + +p = argparse.ArgumentParser() +p.add_argument("--rustc", required=True) +p.add_argument("--tests", required=True) +p.add_argument("--work", required=True) +p.add_argument("--with-rustfix", action="store_true") +p.add_argument("--max", type=int, default=8, help="suggestions tried per test") +p.add_argument("--only") +p.add_argument("--known") +p.add_argument("--jobs", type=int, default=8) +p.add_argument("--pause-on-finding", action="store_true") +p.add_argument("--recheck", action="store_true") +args = p.parse_args() +WORK = Path(args.work).resolve() +(WORK / "scratch").mkdir(parents=True, exist_ok=True) +known = set(Path(args.known).read_text().split()) if args.known else set() + + +def diagnostics(source, flags, edition, out): + """(status, [diagnostic]) for `source`, metadata only.""" + out.mkdir(parents=True, exist_ok=True) + argv = [args.rustc, str(source), "--edition", edition or "2015", "--emit=metadata", "-o", str(out / "x"), + "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", "--error-format=json", + *flags] + try: + r = subprocess.run(argv, capture_output=True, text=True, timeout=120, cwd=out, + env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) + except subprocess.TimeoutExpired: + return "timeout", [] + diags = [] + for line in r.stderr.splitlines(): + try: + diags.append(json.loads(line)) + except json.JSONDecodeError: + pass + if uitest.is_ice(r.stderr): + return "ice", diags + return ("ok" if r.returncode == 0 else "error"), diags + + +def suggestions(diag, file_name): + """The machine-applicable suggestions of a diagnostic, each a list of (start, end, text).""" + out = [] + for child in [diag] + diag.get("children", []): + parts = [(s["byte_start"], s["byte_end"], s["suggested_replacement"]) for s in child.get("spans", []) + if s.get("suggested_replacement") is not None + and s.get("suggestion_applicability") == "MachineApplicable" + and Path(s["file_name"]).name == file_name] + if parts: + out.append(sorted(parts)) + return out + + +def key(diag): + code = (diag.get("code") or {}).get("code") or "" + return code, diag["message"] + + +def primary(diag, file_name): + for s in diag.get("spans", []): + if s.get("is_primary") and Path(s["file_name"]).name == file_name: + return s["byte_start"], s["byte_end"] + return None + + +def errors(diags): + """The errors, by code (or lint name) when they have one: a renamed identifier changes the + message of the same lint.""" + return {(key(d)[0], "" if key(d)[0] else key(d)[1]) for d in diags + if d.get("level") == "error" and not d["message"].startswith("aborting")} + + +def one(path, flags, edition, kind): + rel = str(path.relative_to(args.tests)) + record = {"test": rel} + # The copy is compiled elsewhere: files named by relative path would be missing. + if re.search(r"^\s*(pub(\([^)]*\))?\s+)?mod\s+\w+\s*;|include(_str|_bytes)?!|#\[path", + path.read_text(errors="replace"), re.M): + record["skip"] = "uses files by path" + return record, [] + with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: + d = Path(d) + src = d / path.name + shutil.copy(path, src) + status, diags = diagnostics(src, flags, edition, d / "orig") + if status in ("ice", "timeout"): + record["skip"] = f"original {status}" + return record, [] + base_errors = errors(diags) + text = src.read_bytes() + found, tried = [], 0 + for diag in diags: + for parts in suggestions(diag, path.name): + if tried >= args.max: + break + tried += 1 + # Apply from the end, so earlier offsets stay valid; overlapping parts are skipped. + fixed, last = bytearray(text), None + ok = True + for start, end, repl in sorted(parts, reverse=True): + if last is not None and end > last: + ok = False + break + fixed[start:end] = repl.encode() + last = start + if not ok: + continue + fsrc = d / f"fix{tried}" / path.name + fsrc.parent.mkdir() + fsrc.write_bytes(bytes(fixed)) + fstatus, fdiags = diagnostics(fsrc, flags, edition, fsrc.parent) + what = None + if fstatus == "ice": + what = "ice" + elif any("expected" in m or "unexpected" in m or "unknown start of token" in m + for _, m in errors(fdiags) - base_errors): + what = "parse" + elif diag.get("level") == "warning" and errors(fdiags) - base_errors: + what = "lint-breaks" + elif sum(key(fd) == key(diag) for fd in fdiags) >= sum(key(od) == key(diag) for od in diags): + # Not one fewer of this diagnostic (nested braces legitimately report the next + # level, but there is then one fewer). + what = "not-fixed" + if what: + found.append({"what": what, "diagnostic": diag["message"], "code": key(diag)[0], + "level": diag.get("level"), "parts": parts, + "new_errors": sorted(c or m for c, m in errors(fdiags) - base_errors)[:5], + "fixed_name": f"fix{tried}.rs"}) + shutil.copy(fsrc, d / f"keep-fix{tried}.rs") + record["tried"] = tried + record["found"] = [f"{f['what']}: {f['code'] or f['level']} {f['diagnostic'][:80]}" for f in found] + if found: + out = WORK / "findings" / rel.replace("/", "__") + shutil.rmtree(out, ignore_errors=True) + out.mkdir(parents=True) + shutil.copy(path, out / path.name) + for f in found: + shutil.copy(d / f"keep-{f['fixed_name']}", out / f["fixed_name"]) + (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, + "found": found}, indent=1)) + return record, found + + +def main(): + wanted = None + if args.recheck: + wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} + skip = None if args.with_rustfix else (lambda text, flags: re.search(r"^//@\s*run-rustfix", text, re.M)) + todo = [] + for path, flags, edition, kind in uitest.tests(args.tests, KINDS, skip): + rel = str(path.relative_to(args.tests)) + if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): + continue + todo.append((path, flags, edition, kind)) + print(f"{len(todo)} tests", flush=True) + sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) + + +main() From 3c3caedbf9080f039ff67475c6c6a4c3564f7f59 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 06:16:04 +0000 Subject: [PATCH 02/23] diag-check.py (check 13: debug output in diagnostics, spans out of bounds, errors without a location, duplicates); finding 30 Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- docs/hunt.md | 1 + docs/solver-triage.md | 1 + rustc/diag-check.py | 171 ++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 173 insertions(+) create mode 100644 rustc/diag-check.py diff --git a/docs/hunt.md b/docs/hunt.md index 624a06e..b9dacf8 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -53,6 +53,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 27 | under the new trait solver (nightly's default), a type parameter that appears only in a projection (`T0::Of<'_>`) of a function-pointer coercion is not inferred (E0283); breaks surrealdb through `diskann-wide 0.54.0` | **known, intended**: `diskann-wide` is listed in #160895 ("higher-ranked associated type", the intended breakage of trait-system-refactor-initiative#168; 0.55 not yet patched); surrealdb is an affected project not on that list; found by release-to-release; [facts](hunt/release-regressions.md) | | 28 | glob-import ambiguity depends on item order: with two modules re-exporting each other's globs, one order is E0659 and the other compiles and calls a different function (1.98, nightly); the accepted order has swapped between releases | **looks new**; stable code; found by the equivalent-rewrite differential (`reorder`) on 3 UI tests; [facts](hunt/glob-ambiguity-order.md) | | 29 | machine-applicable lint fixes (what `cargo fix` applies unasked) break builds: `unused_variables` turns `ref b` into a moving `_b` and renames only the declaration of variables mentioned elsewhere, `unused_mut` changes one or-pattern alternative or a variable a `move` closure assigns, `unused_imports` removes a glob that resolution needs | **looks new**; stable 1.98; found by the suggestions-apply check (111 lint fixes in UI tests); six 3–7 line reductions; [facts](hunt/lint-fixes-break-builds.md) | +| 30 | compiler-internal debug output in user-facing diagnostics: under the default (new) solver, E0308 help suggests `as fn(?0t) -> ?0t`; an E0391 cycle note prints `Binder { value: ConstEvaluatable(AliasConst(… DefId(0:7 ~ …` (blessed in `offset-of/inside-array-length.stderr`) | low, diagnostics; found by the diagnostic-invariants check over 18,374 UI tests (excluding tests that ask for verbose output); the first not in CI because of the solver pin ([`solver-triage.md`](solver-triage.md) item I) | Findings 1 and 2 are single-threaded: an ordinary `cargo build`, an edit, another `cargo build`, and the metadata differs from a clean build of the edited source. Both come diff --git a/docs/solver-triage.md b/docs/solver-triage.md index 9e31507..9334218 100644 --- a/docs/solver-triage.md +++ b/docs/solver-triage.md @@ -30,6 +30,7 @@ Sources: the UI-test solver differential ([`solver.md`](solver.md), `rustc/solve | F | higher-ranked associated type no longer guides inference (`escaping-bounds`; `diskann-wide`) | 1 + crate | yes | #160895, tsri#168: intended | low | surrealdb as an affected project on #160895 | | G | type alias `impl Trait`: "does not constrain", a cycle | 2 | no | #160895: RPIT/TAIT handling changed | low | no | | H | `fn_delegation` with `impl Trait` returns: E0282 | 1 | no (incomplete) | no | low | optional | +| I | a help suggestion containing inference variables: "consider casting both fn items to fn pointers using `as fn(?0t) -> ?0t`" (`fn/fn_def_opaque_coercion_to_fn_ptr.rs`; the old solver gives no such help) | 1 | yes | no | low | optional (found by `diag-check.py`) | | 22 | debug assertion `!type_outlives.has_non_rigid_aliases()` in region outlives | 5 | some | sibling of closed #160206 | low | optional: debug builds only, but an invariant broken | | 23 | integer overflow in `ty/instance.rs:421` (`recursion/issue-83150.rs`) | 1 | yes | no | low | optional | diff --git a/rustc/diag-check.py b/rustc/diag-check.py new file mode 100644 index 0000000..de83feb --- /dev/null +++ b/rustc/diag-check.py @@ -0,0 +1,171 @@ +#!/usr/bin/env python3 +"""Diagnostic invariants: what every diagnostic rustc prints must satisfy, whatever the program. + +Compiles each standalone UI test with `--error-format=json` and checks every diagnostic +(children and suggestions included): + + internal user-facing text (message, labels, suggested code) contains compiler-internal + debug output: `DefId(`, region and type-variable debug names (`ReLateParam`, + `ReBound`, `ReVar`, `'{erased}`, `?0t`, `^0`), `Opaque(DefId`, `{closure#0}` in + suggested code + span a span outside its file: byte offsets past the end, lines past the last line, + a start after its end + nowhere an error without any span, other than summaries ("aborting due to") + duplicate the same diagnostic (level, code, message, primary span) twice (noted, not a + finding: some are blessed in .stderr files) + + rustc/diag-check.py --rustc --tests /tests/ui --work [--only ] + [--known ] [--jobs 8] [--pause-on-finding] [--recheck] +""" + +import argparse +import json +import os +import re +import shutil +import subprocess +import sys +import tempfile +from collections import Counter +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parent)) +import uitest # noqa: E402 + +KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) +INTERNAL = re.compile(r"DefId\(|\bRe(LateParam|Bound|Var|Early|Static)\b|'\{erased\}|\?\d+[tif]\b|" + r"'\^\d+(_\d+)?\b|Opaque\(DefId|\bAlias\((Projection|Opaque|Inherent|Free)|" + r"\bBoundRegionKind|\bDefPath\b|\bLocalDefId\b|\bTyKind::") +# In suggested code, compiler-made names that are not Rust. +INTERNAL_CODE = re.compile(r"\{closure#\d+\}|\{opaque#\d+\}|\{async block@|\{impl#\d+\}|\{constant#\d+\}") + +p = argparse.ArgumentParser() +p.add_argument("--rustc", required=True) +p.add_argument("--tests", required=True) +p.add_argument("--work", required=True) +p.add_argument("--only") +p.add_argument("--known") +p.add_argument("--jobs", type=int, default=8) +p.add_argument("--pause-on-finding", action="store_true") +p.add_argument("--recheck", action="store_true") +args = p.parse_args() +WORK = Path(args.work).resolve() +(WORK / "scratch").mkdir(parents=True, exist_ok=True) +known = set(Path(args.known).read_text().split()) if args.known else set() + + +def walk(diag): + yield diag + for c in diag.get("children", []): + yield from walk(c) + + +def one(path, flags, edition, kind): + rel = str(path.relative_to(args.tests)) + record = {"test": rel} + with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: + d = Path(d) + argv = [args.rustc, str(path.resolve()), "--edition", edition or "2015", "--emit=metadata", "-o", + str(d / "x"), "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", + "--error-format=json", *flags] + try: + r = subprocess.run(argv, capture_output=True, text=True, timeout=120, cwd=d, + env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) + except subprocess.TimeoutExpired: + record["skip"] = "timeout" + return record, [] + if uitest.is_ice(r.stderr): + record["skip"] = "ice" + return record, [] + diags = [] + for line in r.stderr.splitlines(): + try: + diags.append(json.loads(line)) + except json.JSONDecodeError: + pass + files = {} + + def file_info(name): + if name not in files: + try: + data = Path(name).read_bytes() if Path(name).is_absolute() else (path.parent / name).read_bytes() + files[name] = (len(data), data.count(b"\n") + 1) + except OSError: + files[name] = None + return files[name] + + found, notes = [], [] + seen = Counter() + for diag in diags: + prim = next(((s["file_name"], s["byte_start"], s["byte_end"]) for s in diag.get("spans", []) + if s.get("is_primary")), None) + seen[(diag.get("level"), (diag.get("code") or {}).get("code"), diag["message"], prim)] += 1 + if (diag.get("level") == "error" and not diag.get("spans") and not diag.get("children") + and not re.match(r"aborting due to|could not compile|\d+ (previous )?errors?", diag["message"]) + and "#![feature" not in diag["message"]): + notes.append({"what": "nowhere", "message": diag["message"][:200]}) + for node in walk(diag): + texts = [node["message"]] + [s.get("label") or "" for s in node.get("spans", [])] + for t in texts: + m = INTERNAL.search(t) + if m: + found.append({"what": "internal", "token": m.group(0), "text": t[:300], + "code": (diag.get("code") or {}).get("code")}) + parts = [] + for s in node.get("spans", []): + repl = s.get("suggested_replacement") + if repl is not None: + m = INTERNAL.search(repl) or INTERNAL_CODE.search(repl) + if m: + found.append({"what": "internal", "token": m.group(0), "text": f"suggests {repl[:200]!r}", + "code": (diag.get("code") or {}).get("code")}) + parts.append((s["file_name"], s["byte_start"], s["byte_end"])) + info = file_info(s["file_name"]) if not s["file_name"].startswith("<") else None + if info: + size, lines = info + if (s["byte_start"] > s["byte_end"] or s["byte_end"] > size or s["line_start"] > lines + or s["line_end"] > lines or s["line_start"] > s["line_end"]): + found.append({"what": "span", "span": {k: s[k] for k in ("file_name", "byte_start", + "byte_end", "line_start", "line_end")}, "size": size, "lines": lines, + "message": node["message"][:200]}) + + for k, n in seen.items(): + if n > 1 and k[0] in ("error", "warning"): + notes.append({"what": "duplicate", "message": k[2][:200], "times": n}) + # One entry per distinct problem. + uniq = {json.dumps(f, sort_keys=True): f for f in found} + found = list(uniq.values()) + record["found"] = [f"{f['what']}: {f.get('token') or ''} {(f.get('text') or f.get('message') or '')[:100]}" for f in found] + record["notes"] = [f"{n['what']}: {n['message'][:100]}" for n in notes] + if found: + out = WORK / "findings" / rel.replace("/", "__") + shutil.rmtree(out, ignore_errors=True) + out.mkdir(parents=True) + shutil.copy(path, out / path.name) + (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, + "found": found, "notes": notes}, indent=1)) + return record, found + + +def main(): + wanted = None + if args.recheck: + wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} + # Tests that ask for compiler internals on purpose: verbose printing, dump attributes. + debug = lambda text, flags: (any(re.search(r"verbose|-Zdump|unpretty|print-", f) for f in flags) + or re.search(r"#!?\[rustc_(dump|effective_visibility|regions|variance|" + r"outlives|layout|abi|def_path|symbol_name|object_lifetime_default|" + r"dump_[a-z_]+|evaluate_where_clauses|then_this_would_need|" + r"if_this_changed|clean|partition)", text) + or "assumptions_on_binders" in text) + todo = [] + for path, flags, edition, kind in uitest.tests(args.tests, KINDS, debug): + rel = str(path.relative_to(args.tests)) + if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): + continue + todo.append((path, flags, edition, kind)) + print(f"{len(todo)} tests", flush=True) + sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) + + +main() From 3ab63d9f5e7d4395e47a4bcabd174f649b3caeb3 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 06:23:11 +0000 Subject: [PATCH 03/23] repro-diff.py (check 15: repeat, path, threads, decoy); nothing new beyond the known parallel-frontend issues Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- docs/checks.md | 8 +++ rustc/repro-diff.py | 146 ++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 154 insertions(+) create mode 100644 rustc/repro-diff.py diff --git a/docs/checks.md b/docs/checks.md index 7027f53..95abcc3 100644 --- a/docs/checks.md +++ b/docs/checks.md @@ -360,3 +360,11 @@ over the standalone UI tests at the pin (and real crates for release-to-release) | release-to-release | `release-diff.py` | 87 real repositories, nightly-2026-07-18 → 10-06 | findings 26 and 27; `allocative` (unstable features) noted | Ten new findings (19–28) in [`hunt.md`](hunt.md), none from the checks mirth had before. + +### Second batch (2026-10-10) + +| check | script | swept | result | +|---|---|---|---| +| suggestions apply (18) | `suggest-diff.py` | 17,945 tests without `run-rustfix`, 7,385 machine-applicable suggestions applied one at a time | finding 29: 111 lint fixes break builds (six shapes reduced); error-recovery suggestions that leave the error or do not parse noted | +| diagnostic invariants (13) | `diag-check.py` | 18,374 tests | finding 30: debug output in two diagnostics; spans all in bounds | +| determinism (15) | `repro-diff.py` | 6,886 tests × repeat, other directory with `--remap-path-prefix`, `-Zthreads=8`, decoy `-L` library | nothing new: only `-Zthreads` differences, all in the known async fn (#162202) and RPIT (#163878) families | diff --git a/rustc/repro-diff.py b/rustc/repro-diff.py new file mode 100644 index 0000000..1770b67 --- /dev/null +++ b/rustc/repro-diff.py @@ -0,0 +1,146 @@ +#!/usr/bin/env python3 +"""Determinism: what rustc writes must depend only on its inputs and options. + +Builds each standalone UI test that compiles (build-pass, run-pass, check-pass) several times +and compares the outputs (`.rmeta`, `.rlib` normalized as artifacts.py does, executables) with +the first build: + + repeat the same build again, in the same directory + path the same build in another directory, both with `--remap-path-prefix` to one name: + what is left of the directory is a path leaking past the remapping + threads `-Zthreads=8` (the parallel front end; tests marked `ignore-parallel-frontend` skip + it); known: async fns (rust-lang/rust#162202), RPIT and impl Trait in traits (#163878) + decoy a `-L` directory holding an unrelated library whose name starts with the crate's + name (rust-lang/rust#159677's shape) + + rustc/repro-diff.py --rustc --tests /tests/ui --work [--variants repeat,path] + [--only ] [--known ] [--jobs 8] [--pause-on-finding] [--recheck] +""" + +import argparse +import hashlib +import json +import os +import re +import shutil +import subprocess +import sys +import tempfile +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parent)) +import artifacts # noqa: E402 +import uitest # noqa: E402 + +VARIANTS = ["repeat", "path", "threads", "decoy"] + +p = argparse.ArgumentParser() +p.add_argument("--rustc", required=True) +p.add_argument("--tests", required=True) +p.add_argument("--work", required=True) +p.add_argument("--variants") +p.add_argument("--only") +p.add_argument("--known") +p.add_argument("--jobs", type=int, default=8) +p.add_argument("--pause-on-finding", action="store_true") +p.add_argument("--recheck", action="store_true") +args = p.parse_args() +WORK = Path(args.work).resolve() +(WORK / "scratch").mkdir(parents=True, exist_ok=True) +variants = args.variants.split(",") if args.variants else VARIANTS +known = set(Path(args.known).read_text().split()) if args.known else set() +OWN = re.compile(r"threads|remap-path|-o\b|--out-dir|emit|crate-name|extern|-L\b|-Cincremental") + + +def build(src_dir, name, flags, edition, kind, extra): + """Compile `src_dir/name`; {output file: digest}, or None if it fails.""" + out = src_dir / "out" + shutil.rmtree(out, ignore_errors=True) + out.mkdir() + emit = "--emit=metadata" if kind == "check-pass" else "--emit=link,metadata" + argv = [args.rustc, name, "--edition", edition or "2015", emit, "--out-dir", "out", "--crate-name", "t", + "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", + f"--remap-path-prefix={src_dir}=/src", *flags, *extra] + try: + r = subprocess.run(argv, capture_output=True, timeout=300, cwd=src_dir, + env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) + except subprocess.TimeoutExpired: + return None + if r.returncode != 0: + return None + files = {} + for f in sorted(out.iterdir()): + if f.suffix == ".rlib": + files[f.name] = artifacts.normalized_rlib(f) + elif f.is_file(): + files[f.name] = hashlib.sha256(f.read_bytes()).hexdigest() + return files + + +def one(path, flags, edition, kind): + rel = str(path.relative_to(args.tests)) + record = {"test": rel} + with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: + d = Path(d) + a, b = d / "a", d / "elsewhere-b" + for x in (a, b): + x.mkdir() + shutil.copy(path, x / path.name) + base = build(a, path.name, flags, edition, kind, []) + if base is None: + record["skip"] = "does not build" + return record, [] + found = [] + text = path.read_text(errors="replace") + for v in variants: + # Tests the parallel front end is known not to handle say so. + if v == "threads" and re.search(r"^//@\s*ignore-parallel-frontend", text, re.M): + continue + if v == "repeat": + got = build(a, path.name, flags, edition, kind, []) + elif v == "path": + got = build(b, path.name, flags, edition, kind, []) + elif v == "threads": + got = build(a, path.name, flags, edition, kind, ["-Zthreads=8"]) + elif v == "decoy": + decoy = d / "decoy" + decoy.mkdir(exist_ok=True) + # An unrelated library sharing the crate's name as a prefix. + (decoy / "libtother.rlib").write_bytes(b"!\n") + (decoy / "libt-0123456789abcdef.rmeta").write_bytes(b"rust\0\0\0\0") + got = build(a, path.name, flags, edition, kind, ["-L", str(decoy)]) + if got is None: + found.append({"variant": v, "what": "does not build"}) + elif got != base: + differ = sorted(k for k in set(got) | set(base) if got.get(k) != base.get(k)) + found.append({"variant": v, "what": "outputs differ: " + ", ".join(differ)}) + # A difference also in `repeat` is nondeterminism of the build itself: report only that. + if any(f["variant"] == "repeat" for f in found): + found = [f for f in found if f["variant"] == "repeat"] + record["found"] = [f"{f['variant']}: {f['what']}" for f in found] + if found: + out = WORK / "findings" / rel.replace("/", "__") + shutil.rmtree(out, ignore_errors=True) + out.mkdir(parents=True) + shutil.copy(path, out / path.name) + (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, + "found": found}, indent=1)) + return record, found + + +def main(): + wanted = None + if args.recheck: + wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} + todo = [] + for path, flags, edition, kind in uitest.tests(args.tests, ("build-pass", "run-pass", "check-pass"), + lambda text, flags: any(OWN.search(f) for f in flags)): + rel = str(path.relative_to(args.tests)) + if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): + continue + todo.append((path, flags, edition, kind)) + print(f"{len(todo)} tests, variants: {', '.join(variants)}", flush=True) + sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) + + +main() From c5fb4b4c4cbafc89766238d22990f10983b940b5 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 06:26:36 +0000 Subject: [PATCH 04/23] gate-check.py (check 17: unstable attributes and library items from stable code); finding 31 Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- docs/checks.md | 1 + docs/hunt.md | 1 + docs/hunt/tests/rustc-main-on-struct.rs | 3 + rustc/gate-check.py | 200 ++++++++++++++++++++++++ 4 files changed, 205 insertions(+) create mode 100644 docs/hunt/tests/rustc-main-on-struct.rs create mode 100644 rustc/gate-check.py diff --git a/docs/checks.md b/docs/checks.md index 95abcc3..a3a362f 100644 --- a/docs/checks.md +++ b/docs/checks.md @@ -368,3 +368,4 @@ Ten new findings (19–28) in [`hunt.md`](hunt.md), none from the checks mirth h | suggestions apply (18) | `suggest-diff.py` | 17,945 tests without `run-rustfix`, 7,385 machine-applicable suggestions applied one at a time | finding 29: 111 lint fixes break builds (six shapes reduced); error-recovery suggestions that leave the error or do not parse noted | | diagnostic invariants (13) | `diag-check.py` | 18,374 tests | finding 30: debug output in two diagnostics; spans all in bounds | | determinism (15) | `repro-diff.py` | 6,886 tests × repeat, other directory with `--remap-path-prefix`, `-Zthreads=8`, decoy `-L` library | nothing new: only `-Zthreads` differences, all in the known async fn (#162202) and RPIT (#163878) families | +| feature gates (17) | `gate-check.py` | 143 unstable attributes × 14 positions; 156 unstable library items with resolvable paths × use, renamed use, glob, impl, value, type | every library spelling gated; finding 31 (an ICE after the gate error for `#[rustc_main]` on non-functions); `#[feature]` outside the crate root only warns (intended) | diff --git a/docs/hunt.md b/docs/hunt.md index b9dacf8..2f0d1c3 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -54,6 +54,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 28 | glob-import ambiguity depends on item order: with two modules re-exporting each other's globs, one order is E0659 and the other compiles and calls a different function (1.98, nightly); the accepted order has swapped between releases | **looks new**; stable code; found by the equivalent-rewrite differential (`reorder`) on 3 UI tests; [facts](hunt/glob-ambiguity-order.md) | | 29 | machine-applicable lint fixes (what `cargo fix` applies unasked) break builds: `unused_variables` turns `ref b` into a moving `_b` and renames only the declaration of variables mentioned elsewhere, `unused_mut` changes one or-pattern alternative or a variable a `move` closure assigns, `unused_imports` removes a glob that resolution needs | **looks new**; stable 1.98; found by the suggestions-apply check (111 lint fixes in UI tests); six 3–7 line reductions; [facts](hunt/lint-fixes-break-builds.md) | | 30 | compiler-internal debug output in user-facing diagnostics: under the default (new) solver, E0308 help suggests `as fn(?0t) -> ?0t`; an E0391 cycle note prints `Binder { value: ConstEvaluatable(AliasConst(… DefId(0:7 ~ …` (blessed in `offset-of/inside-array-length.stderr`) | low, diagnostics; found by the diagnostic-invariants check over 18,374 UI tests (excluding tests that ask for verbose output); the first not in CI because of the solver pin ([`solver-triage.md`](solver-triage.md) item I) | +| 31 | `#[rustc_main]` on a struct, impl, trait or module, on stable: after the expected E0658 and "cannot be used on structs", rustc ICEs ("unexpected sort of node in fn_sig()", `collect.rs`): the item is still taken as the entry point | **looks new**, low (error recovery, internal attribute); regression between 1.91.0 and 1.93.0; found by the feature-gate check; [repro](hunt/tests/rustc-main-on-struct.rs) | Findings 1 and 2 are single-threaded: an ordinary `cargo build`, an edit, another `cargo build`, and the metadata differs from a clean build of the edited source. Both come diff --git a/docs/hunt/tests/rustc-main-on-struct.rs b/docs/hunt/tests/rustc-main-on-struct.rs new file mode 100644 index 0000000..847cc99 --- /dev/null +++ b/docs/hunt/tests/rustc-main-on-struct.rs @@ -0,0 +1,3 @@ +#[rustc_main] +pub struct S; +fn main() {} diff --git a/rustc/gate-check.py b/rustc/gate-check.py new file mode 100644 index 0000000..04cffe6 --- /dev/null +++ b/rustc/gate-check.py @@ -0,0 +1,200 @@ +#!/usr/bin/env python3 +"""Feature gates: nothing unstable may be usable from stable code, whatever the spelling. + +Two enumerations, each compiled without any `#![feature]` and with `RUSTC_BOOTSTRAP` unset +(as a stable user would): + + attributes every attribute in the "Unstable attributes" part of + compiler/rustc_feature/src/builtin_attrs.rs, placed on each kind of item and + position (a function with and without a body, a trait method declaration, a + foreign function, a parameter, a struct, a field, an impl, a module, a closure, a + statement, the crate) + library every top-level `pub` item of core, alloc and std marked `#[unstable(feature)]`, + reached by `use` (directly, renamed, through a glob), implemented (traits) and + taken as a value (functions) + +A program must report the gate (E0658, or "is experimental" / "unstable"). One that compiles is a +finding; one that fails without mentioning the gate is noted (an earlier error may have hidden +it). Library items whose path does not resolve even with the feature are skipped. + + rustc/gate-check.py --rustc --rust --work [--jobs 8] + [--only attributes|library] +""" + +import argparse +import json +import os +import re +import subprocess +import tempfile +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +p = argparse.ArgumentParser() +p.add_argument("--rustc", required=True) +p.add_argument("--rust", required=True) +p.add_argument("--work", required=True) +p.add_argument("--jobs", type=int, default=8) +p.add_argument("--only") +args = p.parse_args() +WORK = Path(args.work).resolve() +WORK.mkdir(parents=True, exist_ok=True) +RUST = Path(args.rust) +ENV = {k: v for k, v in os.environ.items() if k != "RUSTC_BOOTSTRAP"} +GATE = re.compile(r"E0658|is experimental|is unstable|unstable feature|use of unstable|" + r"internal implementation detail|unstable library feature|requires a nightly|" + r"used internally by the standard library|may not be used|are considered unstable|" + r"is an internal|cannot be used on stable") + +# Arguments for attributes that need them; the rest are written bare. +ATTR_ARGS = { + "optimize": "(speed)", "patchable_function_entry": "(prefix_nops = 1, entry_nops = 1)", + "instrument_fn": ' = "on"', "cfi_encoding": ' = "u1x"', "register_tool": "(mytool)", + "register_attribute_tool": "(mytool)", "register_lint_tool": "(mytool)", "linkage": ' = "weak"', + "lang": ' = "mirth_nonexistent"', "rustc_on_unimplemented": '(message = "x")', + "rustc_diagnostic_item": ' = "mirth_x"', "test_runner": "(crate::r)", "pattern_complexity_limit": " = 10", + "rustc_legacy_const_generics": "(0)", "rustc_layout_scalar_valid_range_start": "(1)", + "rustc_abi": "(debug)", "rustc_macro_transparency": ' = "semitransparent"', "unstable": '(feature = "x", issue = "none")', + "stable": '(feature = "x", since = "1.0.0")', "rustc_const_unstable": '(feature = "x", issue = "none")', + "rustc_const_stable": '(feature = "x", since = "1.0.0")', "feature": "(mirth_nonexistent)", + "rustc_objc_class": ' = "X"', "rustc_objc_selector": ' = "x"', "rustc_confusables": '("x")', + "rustc_must_implement_one_of": "(a, b)", "rustc_deprecated_safe_2024": "", + "allow_internal_unstable": "(core_intrinsics)", "rustc_allow_const_fn_unstable": "(x)", + "rustc_default_body_unstable": '(feature = "x", issue = "none")', "unstable_removed": "", + "rustc_simd_monomorphize_lane_limit": ' = "8"', "rustc_scalable_vector": "(4)", +} +# Each position: (name, program with {A} where the attribute goes). +POSITIONS = [ + ("fn", "{A}\npub fn f() {{}}\nfn main() {{}}"), + ("fn-no-body", "pub trait T {{ {A} fn m(&self); }}\nfn main() {{}}"), + ("foreign-fn", 'unsafe extern "C" {{ {A} fn ext(); }}\nfn main() {{}}'), + ("param", "pub fn f({A} x: u32) -> u32 {{ x }}\nfn main() {{}}"), + ("param-no-body", "pub trait T {{ fn m(&self, {A} x: u32); }}\nfn main() {{}}"), + ("fn-ptr-param", "pub type F = fn({A} u32);\nfn main() {{}}"), + ("struct", "{A}\npub struct S;\nfn main() {{}}"), + ("field", "pub struct S {{ {A} pub x: u32 }}\nfn main() {{}}"), + ("impl", "pub struct S;\n{A}\nimpl S {{}}\nfn main() {{}}"), + ("trait", "{A}\npub trait T {{}}\nfn main() {{}}"), + ("mod", "{A}\npub mod m {{}}\nfn main() {{}}"), + ("closure", "fn main() {{ let _c = {A} || (); }}"), + ("statement", "fn main() {{ {A} let _x = 1; }}"), + ("crate", "#![{AI}]\nfn main() {{}}"), +] + + +def compile_text(text, feature=None, crate_type="bin"): + with tempfile.TemporaryDirectory(dir=WORK) as d: + f = Path(d) / "t.rs" + f.write_text((f"#![feature({feature})]\n" if feature else "") + text) + env = dict(ENV, RUSTC_BOOTSTRAP="1") if feature else ENV + try: + r = subprocess.run([args.rustc, str(f), "--edition", "2021", "--emit=metadata", "--crate-type", + crate_type, "-o", str(Path(d) / "x")], capture_output=True, text=True, + timeout=60, cwd=d, env=env) + except subprocess.TimeoutExpired: + return "timeout", "" + if "internal compiler error" in r.stderr or "panicked" in r.stderr: + return "ice", r.stderr + return ("ok" if r.returncode == 0 else "error"), r.stderr + + +def classify(status, stderr, item=None): + # `#[feature]` outside the crate root does nothing; rustc warns that it belongs at the root. + if status == "ok" and item == "feature" and "crate-level attribute" in stderr: + return "gated" + if status == "ok": + return "accepted" + if status == "ice": + return "ice" + return "gated" if GATE.search(stderr) else "no-gate-message" + + +def attributes(): + text = (RUST / "compiler/rustc_feature/src/builtin_attrs.rs").read_text() + names = sorted(set(re.findall(r"sym::([a-z_0-9]+)", text[text.index("Unstable attributes:"):]))) + jobs = [] + for name in names: + arg = ATTR_ARGS.get(name, "") + for pos, template in POSITIONS: + jobs.append((name, pos, template.replace("{AI}", f"{name}{arg}").replace("{A}", f"#[{name}{arg}]") + .replace("{{", "{").replace("}}", "}"))) + + def run(job): + name, pos, prog = job + status, err = compile_text(prog) + return {"kind": "attribute", "item": name, "position": pos, "result": classify(status, err, name), + "first": next((l for l in err.splitlines() if l.startswith("error")), "")[:200]} + with ThreadPoolExecutor(args.jobs) as ex: + return list(ex.map(run, jobs)) + + +def library_items(): + items = [] + for krate in ("core", "alloc", "std"): + root = RUST / "library" / krate / "src" + for f in root.rglob("*.rs"): + rel = f.relative_to(root).with_suffix("") + parts = [x for x in rel.parts if x not in ("lib", "mod")] + module = "::".join([krate] + parts) + lines = f.read_text(errors="replace").splitlines() + for i, line in enumerate(lines): + m = re.match(r'#\[unstable\(feature = "([a-z_0-9]+)"', line) + if not m: + continue + for nxt in lines[i + 1:i + 6]: + if nxt.startswith("#"): + continue + d = re.match(r"pub (?:const |unsafe |auto |extern \"C\" )*(struct|enum|trait|union|type|fn|const|static|macro) ([A-Za-z_][A-Za-z_0-9]*)(<)?", nxt) + if d: + items.append({"crate": krate, "path": f"{module}::{d.group(2)}", "kind": d.group(1), + "feature": m.group(1), "generic": bool(d.group(3)), + "file": str(f.relative_to(RUST))}) + break + return items + + +def library(): + items = library_items() + + def run(item): + path, feature, kind = item["path"], item["feature"], item["kind"] + # The path must resolve with the feature on, or the item is not where its file says. + status, _ = compile_text(f"#[allow(unused_imports)] use {path};\nfn main() {{}}", feature) + if status != "ok": + return [{"kind": "library", "item": path, "position": "path", "result": "skipped: path"}] + parent, name = path.rsplit("::", 1) + progs = [("use", f"#[allow(unused_imports)] use {path};\nfn main() {{}}"), + ("use-as", f"#[allow(unused_imports)] use {path} as Renamed;\nfn main() {{}}"), + ("glob", f"#[allow(unused_imports)] use {parent}::*;\n#[allow(unused_imports)] use self::{name} as _;\nfn main() {{}}")] + if kind == "trait" and not item["generic"]: + progs.append(("impl", f"struct L;\nimpl {path} for L {{}}\nfn main() {{}}")) + if kind == "fn" and not item["generic"]: + progs.append(("value", f"fn main() {{ let _f = {path}; }}")) + if kind in ("struct", "enum", "union", "type") and not item["generic"]: + progs.append(("type", f"pub fn g(_: Option<&{path}>) {{}}\nfn main() {{}}")) + out = [] + for pos, prog in progs: + status, err = compile_text(prog) + out.append({"kind": "library", "item": path, "feature": feature, "position": pos, + "result": classify(status, err), + "first": next((l for l in err.splitlines() if l.startswith("error")), "")[:200]}) + return out + with ThreadPoolExecutor(args.jobs) as ex: + return [r for rs in ex.map(run, items) for r in rs] + + +def main(): + results = [] + if args.only in (None, "attributes"): + results += attributes() + if args.only in (None, "library"): + results += library() + (WORK / "results.json").write_text(json.dumps(results, indent=0)) + from collections import Counter + print(Counter((r["kind"], r["result"]) for r in results)) + for r in results: + if r["result"] in ("accepted", "ice"): + print(f"{r['result'].upper():9} {r['kind']:9} {r['item']:45} {r['position']}") + + +main() From 761747df325b7d2eaf7fe605844457b9e8db3593 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 06:51:35 +0000 Subject: [PATCH 05/23] instr-check.py (check 21: PGO and coverage round trips), xlink.py + xlink-probe (check 5: build-std and link for every target) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- rustc/instr-check.py | 162 ++++++++++++++++++++++++++++++++++ rustc/xlink-probe/Cargo.lock | 7 ++ rustc/xlink-probe/Cargo.toml | 10 +++ rustc/xlink-probe/src/main.rs | 39 ++++++++ rustc/xlink.py | 123 ++++++++++++++++++++++++++ 5 files changed, 341 insertions(+) create mode 100644 rustc/instr-check.py create mode 100644 rustc/xlink-probe/Cargo.lock create mode 100644 rustc/xlink-probe/Cargo.toml create mode 100644 rustc/xlink-probe/src/main.rs create mode 100644 rustc/xlink.py diff --git a/rustc/instr-check.py b/rustc/instr-check.py new file mode 100644 index 0000000..f4521ce --- /dev/null +++ b/rustc/instr-check.py @@ -0,0 +1,162 @@ +#!/usr/bin/env python3 +"""Instrumentation round trip: instrumented programs must behave as uninstrumented ones and write +profiles that LLVM's tools accept. + +For each runnable UI test (run-pass), with a toolchain that ships the profiler runtime and +llvm-tools (the pinned nightly): + + pgo build with `-Cprofile-generate`, run, `llvm-profdata merge` the raw profile, rebuild + with `-Cprofile-use`, run again + coverage build with `-Cinstrument-coverage`, run, merge, `llvm-cov export` the binary + +Findings: an instrumented or profile-guided program whose exit status or stdout differs from the +plain build; no raw profile written; `llvm-profdata` or `llvm-cov` failing or crashing, or +warning about corrupt or malformed data; the compiler crashing on `-Cprofile-use`. + + rustc/instr-check.py --toolchain nightly-2026-10-06 --tests /tests/ui --work + [--only ] [--limit N] [--jobs 8] [--pause-on-finding] [--recheck] +""" + +import argparse +import json +import os +import re +import shutil +import subprocess +import sys +import tempfile +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parent)) +import uitest # noqa: E402 + +p = argparse.ArgumentParser() +p.add_argument("--toolchain", required=True) +p.add_argument("--tests", required=True) +p.add_argument("--work", required=True) +p.add_argument("--only") +p.add_argument("--limit", type=int) +p.add_argument("--jobs", type=int, default=8) +p.add_argument("--pause-on-finding", action="store_true") +p.add_argument("--recheck", action="store_true") +args = p.parse_args() +WORK = Path(args.work).resolve() +(WORK / "scratch").mkdir(parents=True, exist_ok=True) +TC = Path.home() / f".rustup/toolchains/{args.toolchain}-x86_64-unknown-linux-gnu" +RUSTC = TC / "bin/rustc" +TOOLS = TC / "lib/rustlib/x86_64-unknown-linux-gnu/bin" +OWN = re.compile(r"profile|instrument-coverage|coverage-options|-O$|opt-level|panic=|prefer-dynamic|" + r"codegen-backend|-Clto|lto=|no-prepopulate") +BAD = re.compile(r"corrupt|malformed|invalid|truncated|failed to|error", re.I) + + +def tool(name, *a, cwd=None): + try: + r = subprocess.run([str(TOOLS / name), *a], capture_output=True, text=True, timeout=120, cwd=cwd) + return r.returncode, r.stderr + r.stdout[-200:] + except subprocess.TimeoutExpired: + return "timeout", "" + + +def observe(binary, env=None): + try: + r = subprocess.run([str(binary)], capture_output=True, timeout=30, cwd=binary.parent, + stdin=subprocess.DEVNULL, env=dict(os.environ, RUST_BACKTRACE="0", **(env or {}))) + except subprocess.TimeoutExpired: + return ("timeout", "") + code = r.returncode if r.returncode >= 0 else f"signal {-r.returncode}" + return (code, r.stdout.decode(errors="replace")) + + +def one(path, flags, edition, kind): + rel = str(path.relative_to(args.tests)) + record = {"test": rel} + found = [] + with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: + d = Path(d) + status, _, plain = uitest.compile(RUSTC, path.resolve(), d / "plain", flags, edition, ["-Copt-level=1"]) + if plain is None: + record["skip"] = f"plain build {status}" + return record, [] + base = observe(plain) + if base != observe(plain): + record["skip"] = "nondeterministic" + return record, [] + # PGO: generate, merge, use. + raw = d / "pgo-raw" + status, err, gen = uitest.compile(RUSTC, path.resolve(), d / "gen", flags, edition, + ["-Copt-level=1", f"-Cprofile-generate={raw}"]) + if gen is None: + found.append({"what": f"profile-generate build {status}", "stderr": err[-1500:]}) + else: + got = observe(gen) + if got != base: + found.append({"what": "profile-generate changes behavior", "base": base, "got": got}) + profraws = list(raw.glob("*.profraw")) if raw.exists() else [] + if not profraws and got[0] == 0: + found.append({"what": "no raw profile written"}) + elif profraws: + merged = d / "merged.profdata" + code, out = tool("llvm-profdata", "merge", "-o", str(merged), *map(str, profraws)) + if code != 0 or BAD.search(out): + found.append({"what": "llvm-profdata merge", "code": code, "out": out[-1500:]}) + else: + status, err, use = uitest.compile(RUSTC, path.resolve(), d / "use", flags, edition, + ["-Copt-level=2", f"-Cprofile-use={merged}"]) + if status == "ice": + found.append({"what": "profile-use ICE", "stderr": err[-2000:]}) + elif use is None: + found.append({"what": f"profile-use build {status}", "stderr": err[-1500:]}) + elif observe(use) != base: + found.append({"what": "profile-use changes behavior", "base": base, "got": observe(use)}) + # Coverage: instrument, merge, export. + status, err, cov = uitest.compile(RUSTC, path.resolve(), d / "cov", flags, edition, ["-Cinstrument-coverage"]) + if cov is None: + found.append({"what": f"instrument-coverage build {status}", "stderr": err[-1500:]}) + else: + craw = d / "cov-raw" + craw.mkdir() + got = observe(cov, {"LLVM_PROFILE_FILE": str(craw / "c-%p.profraw")}) + if got != base: + found.append({"what": "instrument-coverage changes behavior", "base": base, "got": got}) + profraws = list(craw.glob("*.profraw")) + if profraws: + merged = d / "cov.profdata" + code, out = tool("llvm-profdata", "merge", "-sparse", "-o", str(merged), *map(str, profraws)) + if code != 0 or BAD.search(out): + found.append({"what": "llvm-profdata merge (coverage)", "code": code, "out": out[-1500:]}) + else: + code, out = tool("llvm-cov", "export", "-summary-only", f"-instr-profile={merged}", str(cov)) + if code != 0: + found.append({"what": "llvm-cov export", "code": code, "out": out[-1500:]}) + elif got[0] == 0: + found.append({"what": "no coverage profile written"}) + record["found"] = [f["what"] for f in found] + if found: + out = WORK / "findings" / rel.replace("/", "__") + shutil.rmtree(out, ignore_errors=True) + out.mkdir(parents=True) + shutil.copy(path, out / path.name) + (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, + "found": found}, indent=1, default=str)) + return record, found + + +def main(): + wanted = None + if args.recheck: + wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} + todo = [] + for path, flags, edition, kind in uitest.tests(args.tests, ("run-pass",), + lambda text, flags: any(OWN.search(f) for f in flags)): + rel = str(path.relative_to(args.tests)) + if (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): + continue + todo.append((path, flags, edition, kind)) + if args.limit: + todo = todo[:args.limit] + print(f"{len(todo)} tests", flush=True) + sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) + + +main() diff --git a/rustc/xlink-probe/Cargo.lock b/rustc/xlink-probe/Cargo.lock new file mode 100644 index 0000000..30e7b5f --- /dev/null +++ b/rustc/xlink-probe/Cargo.lock @@ -0,0 +1,7 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "probe" +version = "0.1.0" diff --git a/rustc/xlink-probe/Cargo.toml b/rustc/xlink-probe/Cargo.toml new file mode 100644 index 0000000..61d6a70 --- /dev/null +++ b/rustc/xlink-probe/Cargo.toml @@ -0,0 +1,10 @@ +[package] +name = "probe" +version = "0.1.0" +edition = "2021" +[profile.dev] +panic = "abort" +[profile.release] +panic = "abort" + +[workspace] diff --git a/rustc/xlink-probe/src/main.rs b/rustc/xlink-probe/src/main.rs new file mode 100644 index 0000000..298d0d4 --- /dev/null +++ b/rustc/xlink-probe/src/main.rs @@ -0,0 +1,39 @@ +#![no_std] +#![no_main] +#![feature(core_float_math)] +use core::fmt::Write; +use core::hint::black_box; + +struct Buf { b: [u8; 256], n: usize } +impl Write for Buf { + fn write_str(&mut self, s: &str) -> core::fmt::Result { + for &c in s.as_bytes() { if self.n < 256 { self.b[self.n] = c; self.n += 1; } } + Ok(()) + } +} + +#[unsafe(no_mangle)] +pub extern "C" fn probe_entry() -> u32 { + let a: u128 = black_box(0x1234_5678_9abc_def0_1122_3344_5566_7788); + let b: u128 = black_box(12345); + let i: i128 = black_box(-987654321987654321i128); + let f: f64 = black_box(1.5e10); + let g: f32 = black_box(2.5); + let mut buf = Buf { b: [0; 256], n: 0 }; + let _ = write!(buf, "{} {} {} {:.3} {:e} {}", a / b, a % b, i / 7, f, g as f64, (f as i128) as f32); + let big = black_box([7u8; 4096]); + let mut dst = [0u8; 4096]; + dst.copy_from_slice(&big); + let m = core::f64::math::mul_add(f, black_box(1.25), black_box(3.0)) + core::f64::math::sqrt(f) + core::f64::math::floor(f) + core::f32::math::mul_add(g, g, g) as f64; + let x = (m as u64) + black_box(f) as u64 + black_box(g) as u64 + (i as f64) as u64; + #[cfg(target_has_atomic = "32")] + { + use core::sync::atomic::{AtomicU32, Ordering}; + static C: AtomicU32 = AtomicU32::new(0); + C.fetch_add(1, Ordering::SeqCst); + } + (buf.n as u32) ^ (dst[100] as u32) ^ (x as u32) +} + +#[panic_handler] +fn panic(_: &core::panic::PanicInfo) -> ! { loop {} } diff --git a/rustc/xlink.py b/rustc/xlink.py new file mode 100644 index 0000000..fab08ac --- /dev/null +++ b/rustc/xlink.py @@ -0,0 +1,123 @@ +#!/usr/bin/env python3 +"""Cross-target build and link: every target rustc knows must build `core` and `alloc` and link a +program with its documented linker, with no undefined symbols. + +For each target, builds rustc/xlink-probe (a `no_std` program using 128-bit integers, float +conversions and math, float formatting, large copies and atomics, so that it needs the +compiler-builtins routines that go missing) with `cargo -Zbuild-std=core,alloc` and links it. +Targets whose default linker is a C compiler driver (absent for cross targets here) link with +`rust-lld` in the flavor their spec names instead. lld reports undefined symbols as errors, which +is what this looks for. + + link-undefined the link fails with undefined symbols + link the link fails otherwise + env a library or startup file of the target's C sysroot is missing here + build core or alloc (or the probe) does not compile for the target + ice the compiler crashes + + rustc/xlink.py --toolchain nightly-2026-10-06 --work [--targets t1,t2] [--jobs 6] + +Writes /results.json. Target directories are removed after each build (about 100 MB each). +""" + +import argparse +import json +import os +import re +import shutil +import subprocess +from collections import Counter +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +p = argparse.ArgumentParser() +p.add_argument("--toolchain", required=True) +p.add_argument("--work", required=True) +p.add_argument("--targets") +p.add_argument("--jobs", type=int, default=6) +args = p.parse_args() +WORK = Path(args.work).resolve() +WORK.mkdir(parents=True, exist_ok=True) +PROBE = Path(__file__).resolve().parent / "xlink-probe" +# Targets that need more than a target name (a CPU, an external linker that is not lld). +SKIP = re.compile(r"^(amdgcn|nvptx|bpf|spirv)|-uefi-|avr-none") + + +def rustc(*a): + return subprocess.run(["rustc", f"+{args.toolchain}", *a], capture_output=True, text=True, + env=dict(os.environ, RUSTC_BOOTSTRAP="1")) + + +def link_flags(spec): + """RUSTFLAGS for linking the probe with lld, from the target's spec.""" + flavor = spec.get("linker-flavor", "") + linker = spec.get("linker", "") + own_lld = "lld" in linker or flavor.endswith("-lld") or flavor in ("wasm-lld", "wasm-lld-cc") + flags = [] + if spec.get("is-like-wasm") or flavor.startswith("wasm"): + return flags + ["-Clink-arg=--no-entry", "-Clink-arg=--export=probe_entry"] + if flavor.startswith("msvc") or spec.get("is-like-msvc"): + if not own_lld: + flags += ["-Clinker=rust-lld", "-Clinker-flavor=lld-link"] + return flags + ["-Clink-arg=/ENTRY:probe_entry", "-Clink-arg=/NODEFAULTLIB"] + if flavor.startswith("darwin") or spec.get("is-like-darwin"): + if not own_lld: + flags += ["-Clinker=rust-lld", "-Clinker-flavor=ld64.lld"] + return flags + ["-Clink-arg=-e", "-Clink-arg=_probe_entry", "-Clink-arg=-undefined", + "-Clink-arg=dynamic_lookup"] + if not own_lld: + flags += ["-Clinker=rust-lld", "-Clinker-flavor=ld.lld"] + return flags + ["-Clink-arg=--entry=probe_entry"] + + +def one(target): + if SKIP.search(target): + return target, {"result": "skipped"} + out = rustc("--print", "target-spec-json", "-Zunstable-options", "--target", target) + if out.returncode != 0: + return target, {"result": "skipped", "why": "no spec"} + spec = json.loads(out.stdout) + flags = link_flags(spec) + tdir = WORK / "target" / target + env = dict(os.environ, CARGO_TARGET_DIR=str(tdir), RUSTFLAGS=" ".join(flags), CARGO_TERM_COLOR="never") + env.pop("RUSTC_WRAPPER", None) + try: + r = subprocess.run(["cargo", f"+{args.toolchain}", "build", "--release", "-Zbuild-std=core,alloc", + "-Zbuild-std-features=compiler-builtins-mem", "--target", target], + cwd=PROBE, env=env, capture_output=True, text=True, timeout=1500) + code, err = r.returncode, r.stderr + except subprocess.TimeoutExpired: + code, err = "timeout", "" + shutil.rmtree(tdir, ignore_errors=True) + if code == 0: + result = "ok" + elif "internal compiler error" in err or "panicked at" in err: + result = "ice" + elif "linking with" in err or "rust-lld: error" in err or "lld: error" in err: + if re.search(r"undefined (symbol|reference)", err): + result = "link-undefined" + elif re.search(r"unable to find library|cannot open crt|cannot open .*\.o\b|No such file", err): + # Libraries or startup files that ship with the target's prebuilt std or a C sysroot. + result = "env" + else: + result = "link" + else: + result = "build" + undefined = sorted(set(re.findall(r"undefined symbol: ([^\s\n]+)", err)))[:20] + first = next((l for l in err.splitlines() if re.search(r"error(\[|:)", l)), "")[:300] + return target, {"result": result, "flags": flags, "undefined": undefined, "first": first, + "tail": err[-2500:] if code != 0 else ""} + + +def main(): + targets = args.targets.split(",") if args.targets else rustc("--print", "target-list").stdout.split() + with ThreadPoolExecutor(args.jobs) as ex: + results = dict(ex.map(one, targets)) + (WORK / "results.json").write_text(json.dumps(results, indent=1)) + print(Counter(r["result"] for r in results.values())) + for t, r in sorted(results.items()): + if r["result"] not in ("ok", "skipped"): + print(f"{r['result']:15} {t:40} {' '.join(r['undefined'][:6]) or r['first'][:120]}") + + +main() From 72ec92cabf5e9e39d2327151b808e7c959973272 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 06:53:30 +0000 Subject: [PATCH 06/23] scale-check.py (check 10: growth exponents of compile time, memory, future size and stack frames over size-parameterized programs) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- rustc/scale-check.py | 194 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 194 insertions(+) create mode 100644 rustc/scale-check.py diff --git a/rustc/scale-check.py b/rustc/scale-check.py new file mode 100644 index 0000000..924fc86 --- /dev/null +++ b/rustc/scale-check.py @@ -0,0 +1,194 @@ +#!/usr/bin/env python3 +"""Scaling and budgets: compile time, memory, future sizes and stack frames must grow at most +about linearly with the size of a program of a fixed shape. + +Each generator writes a program of size N for N in a doubling series. For each N the check records +the compiler's user CPU time and peak memory (wait4), and for some shapes a size the program itself +reports (`size_of_val` of a future) or the largest stack frame in the assembly. It fits the growth +exponent k in value ~ N^k from the largest sizes (log-log slope) and reports: + + superlinear k above --max-exponent (default 1.6) for time or memory, or above 1.3 for a size + timeout a compile over --timeout seconds + + rustc/scale-check.py --rustc --work [--only fields,enum] [--opt 0,2] + [--sizes 50,100,200,400,800] [--timeout 300] [--max-exponent 1.6] [--jobs 4] +""" + +import argparse +import json +import math +import os +import re +import subprocess +import tempfile +import time +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +p = argparse.ArgumentParser() +p.add_argument("--rustc", required=True) +p.add_argument("--work", required=True) +p.add_argument("--only") +p.add_argument("--opt", default="0,2") +p.add_argument("--sizes", default="50,100,200,400,800") +p.add_argument("--timeout", type=int, default=300) +p.add_argument("--max-exponent", type=float, default=1.6) +p.add_argument("--jobs", type=int, default=4) +args = p.parse_args() +WORK = Path(args.work).resolve() +WORK.mkdir(parents=True, exist_ok=True) + + +def g_fields(n): + fields = "\n".join(f" pub f{i}: u{8 << (i % 4)}," for i in range(n)) + return f"#[derive(Debug, Clone, PartialEq, Eq, Hash, Default, PartialOrd, Ord)]\npub struct S {{\n{fields}\n}}\nfn main() {{ let s = S::default(); println!(\"{{}}\", format!(\"{{:?}}\", s.clone()).len()); }}\n" + + +def g_enum(n): + variants = "\n".join(f" V{i}(u32)," for i in range(n)) + arms = "\n".join(f" E::V{i}(x) => x + {i}," for i in range(n)) + return f"#[derive(Debug, Clone, PartialEq)]\npub enum E {{\n{variants}\n}}\npub fn f(e: &E) -> u32 {{\n match *e {{\n{arms}\n }}\n}}\nfn main() {{ println!(\"{{}}\", f(&E::V0(1))); }}\n" + + +def g_nested_generic(n): + ty = "u8" + for _ in range(n): + ty = f"W<{ty}>" + return ("#[derive(Clone, Debug, Default)] pub struct W(T);\n" + "pub trait T { fn t(&self) -> usize; }\nimpl T for u8 { fn t(&self) -> usize { 1 } }\n" + "impl T for W { fn t(&self) -> usize { self.0.t() + 1 } }\n" + f"fn main() {{ let v: {ty} = Default::default(); println!(\"{{}}\", v.t()); }}\n") + + +def g_iter_chain(n): + chain = "".join(f".map(|x| x.wrapping_add({i}))" for i in range(n)) + return f"fn main() {{ let s: u64 = (0u64..10){chain}.sum(); println!(\"{{}}\", s); }}\n" + + +def g_async_forward(n): + fns = ["async fn f0(x: [u8; 64]) -> u8 { x[0] }"] + for i in range(1, n): + fns.append(f"async fn f{i}(x: [u8; 64]) -> u8 {{ f{i - 1}(x).await }}") + return ("\n".join(fns) + f"\nfn main() {{ let fut = f{n - 1}([1; 64]); " + "println!(\"SIZE {}\", std::mem::size_of_val(&fut)); }\n") + + +def g_seq_calls(n): + calls = "\n".join(f" let a{i} = big({i}); acc ^= a{i}[{i % 512}];" for i in range(n)) + return ("#[inline(never)] fn big(x: u64) -> [u64; 512] { [x; 512] }\n" + f"#[inline(never)] pub fn many() -> u64 {{\n let mut acc = 0u64;\n{calls}\n acc\n}}\n" + "fn main() { println!(\"{}\", many()); }\n") + + +def g_trait_impls(n): + impls = "\n".join(f"pub struct S{i}; impl Tr for S{i} {{ fn v(&self) -> u32 {{ {i} }} }}" for i in range(n)) + uses = " + ".join(f"S{i}.v()" for i in range(n)) + return f"pub trait Tr {{ fn v(&self) -> u32; }}\n{impls}\nfn main() {{ println!(\"{{}}\", {uses}); }}\n" + + +def g_nested_expr(n): + expr = "1u64" + for i in range(n): + expr = f"({expr} + {i % 7})" + return f"fn main() {{ let x = std::hint::black_box({expr}); println!(\"{{}}\", x); }}\n" + + +GENERATORS = { + "fields": (g_fields, None), "enum": (g_enum, None), "nested-generic": (g_nested_generic, None), + "iter-chain": (g_iter_chain, None), "async-forward": (g_async_forward, "run-size"), + "seq-calls": (g_seq_calls, "frame"), "trait-impls": (g_trait_impls, None), "nested-expr": (g_nested_expr, None), +} +# Shapes where a size of N cannot be compiled meaningfully past a bound (recursion limits). +CAP = {"nested-generic": 120, "nested-expr": 400, "async-forward": 200} + + +def measure(source, opt): + with tempfile.TemporaryDirectory(dir=WORK) as d: + d = Path(d) + (d / "m.rs").write_text(source) + argv = [args.rustc, "m.rs", "--edition", "2021", f"-Copt-level={opt}", "-o", "m", "--emit=link,asm"] + start = time.time() + proc = subprocess.Popen(argv, cwd=d, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE) + deadline = start + args.timeout + while True: + pid, status, usage = os.wait4(proc.pid, os.WNOHANG) + if pid: + break + if time.time() > deadline: + proc.kill() + os.wait4(proc.pid, 0) + return {"timeout": True} + time.sleep(0.05) + err = proc.stderr.read().decode(errors="replace") + if os.waitstatus_to_exitcode(status) != 0: + return {"error": err[-500:]} + out = {"user": usage.ru_utime, "rss_kb": usage.ru_maxrss, "wall": time.time() - start} + asm = (d / "m.s").read_text(errors="replace") + frames = [int(x, 16) if x.startswith("0x") else int(x) + for x in re.findall(r"sub[q]?\s+\$(0x[0-9a-f]+|\d+),\s*%rsp", asm)] + out["max_frame"] = max(frames) if frames else 0 + try: + run = subprocess.run([str(d / "m")], capture_output=True, text=True, timeout=30) + m = re.search(r"SIZE (\d+)", run.stdout) + if m: + out["run_size"] = int(m.group(1)) + except subprocess.TimeoutExpired: + pass + return out + + +def exponent(ns, values): + pts = [(math.log(n), math.log(v)) for n, v in zip(ns, values) if v and v > 0] + if len(pts) < 3: + return None + pts = pts[-3:] # the largest sizes: constant overheads dominate the small ones + mx = sum(x for x, _ in pts) / len(pts) + my = sum(y for _, y in pts) / len(pts) + den = sum((x - mx) ** 2 for x, _ in pts) + return sum((x - mx) * (y - my) for x, y in pts) / den if den else None + + +def one(job): + name, opt = job + gen, extra = GENERATORS[name] + sizes = [n for n in map(int, args.sizes.split(",")) if n <= CAP.get(name, 10**9)] + rows = [] + for n in sizes: + r = measure(gen(n), opt) + r["n"] = n + rows.append(r) + if r.get("timeout") or r.get("error"): + break + ok = [r for r in rows if "user" in r] + ns = [r["n"] for r in ok] + result = {"shape": name, "opt": opt, "rows": rows, + "k_time": exponent(ns, [max(r["user"] - ok[0]["user"] * 0.5, 1e-3) for r in ok]) if ok else None, + "k_rss": exponent(ns, [r["rss_kb"] for r in ok]), + "k_frame": exponent(ns, [r["max_frame"] for r in ok]) if extra == "frame" else None, + "k_size": exponent(ns, [r.get("run_size", 0) for r in ok]) if extra == "run-size" else None} + found = [] + if any(r.get("timeout") for r in rows): + found.append(f"timeout at N={rows[-1]['n']}") + if any(r.get("error") for r in rows): + found.append(f"error at N={rows[-1]['n']}: {rows[-1]['error'][-200:]}") + for key, limit in (("k_time", args.max_exponent), ("k_rss", args.max_exponent), ("k_frame", 1.3), ("k_size", 1.3)): + if result[key] is not None and result[key] > limit: + found.append(f"{key} = {result[key]:.2f}") + result["found"] = found + return result + + +def main(): + names = args.only.split(",") if args.only else list(GENERATORS) + jobs = [(n, int(o)) for n in names for o in args.opt.split(",")] + with ThreadPoolExecutor(args.jobs) as ex: + results = list(ex.map(one, jobs)) + (WORK / "results.json").write_text(json.dumps(results, indent=1)) + for r in results: + last = next((x for x in reversed(r["rows"]) if "user" in x), {}) + ks = " ".join(f"{k}={r[k]:.2f}" for k in ("k_time", "k_rss", "k_frame", "k_size") if r[k] is not None) + print(f"{r['shape']:15} O{r['opt']} N<={last.get('n')} user {last.get('user', 0):.1f}s " + f"rss {last.get('rss_kb', 0) // 1024}MB {ks} {'; '.join(r['found'])}") + + +main() From cfcebd3fb921f70b4748e8fab7df44672003e476 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 07:16:05 +0000 Subject: [PATCH 07/23] mirth-lab: the checks in Rust. Core library (uitest, rustc with typed JSON diagnostics and timeouts, driver with the frontier options, artifacts, miri, normalize) and opt-diff, solver-diff, crash-diff, diag-check, validated against the Python sweeps (same findings); mirth-rewrite split into a library and a binary Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- Cargo.lock | 437 ++++++++++++++++++++++ Cargo.toml | 1 + crates/mirth-lab/Cargo.toml | 24 ++ crates/mirth-lab/src/artifacts.rs | 105 ++++++ crates/mirth-lab/src/driver.rs | 161 ++++++++ crates/mirth-lab/src/lib.rs | 9 + crates/mirth-lab/src/main.rs | 49 +++ crates/mirth-lab/src/miri.rs | 82 ++++ crates/mirth-lab/src/normalize.rs | 49 +++ crates/mirth-lab/src/rustc.rs | 286 ++++++++++++++ crates/mirth-lab/src/tools/crash_diff.rs | 100 +++++ crates/mirth-lab/src/tools/diag_check.rs | 144 +++++++ crates/mirth-lab/src/tools/opt_diff.rs | 185 +++++++++ crates/mirth-lab/src/tools/solver_diff.rs | 138 +++++++ crates/mirth-lab/src/uitest.rs | 182 +++++++++ crates/mirth-rewrite/src/lib.rs | 319 ++++++++++++++++ crates/mirth-rewrite/src/main.rs | 314 +--------------- 17 files changed, 2282 insertions(+), 303 deletions(-) create mode 100644 crates/mirth-lab/Cargo.toml create mode 100644 crates/mirth-lab/src/artifacts.rs create mode 100644 crates/mirth-lab/src/driver.rs create mode 100644 crates/mirth-lab/src/lib.rs create mode 100644 crates/mirth-lab/src/main.rs create mode 100644 crates/mirth-lab/src/miri.rs create mode 100644 crates/mirth-lab/src/normalize.rs create mode 100644 crates/mirth-lab/src/rustc.rs create mode 100644 crates/mirth-lab/src/tools/crash_diff.rs create mode 100644 crates/mirth-lab/src/tools/diag_check.rs create mode 100644 crates/mirth-lab/src/tools/opt_diff.rs create mode 100644 crates/mirth-lab/src/tools/solver_diff.rs create mode 100644 crates/mirth-lab/src/uitest.rs create mode 100644 crates/mirth-rewrite/src/lib.rs diff --git a/Cargo.lock b/Cargo.lock index e285699..3851a8e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -11,6 +11,129 @@ dependencies = [ "memchr", ] +[[package]] +name = "anstream" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "824a212faf96e9acacdbd09febd34438f8f711fb84e09a8916013cd7815ca28d" +dependencies = [ + "anstyle", + "anstyle-parse", + "anstyle-query", + "anstyle-wincon", + "colorchoice", + "is_terminal_polyfill", + "utf8parse", +] + +[[package]] +name = "anstyle" +version = "1.0.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "940b3a0ca603d1eade50a4846a2afffd5ef57a9feac2c0e2ec2e14f9ead76000" + +[[package]] +name = "anstyle-parse" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52ce7f38b242319f7cabaa6813055467063ecdc9d355bbb4ce0c68908cd8130e" +dependencies = [ + "utf8parse", +] + +[[package]] +name = "anstyle-query" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "40c48f72fd53cd289104fc64099abca73db4166ad86ea0b4341abe65af83dadc" +dependencies = [ + "windows-sys", +] + +[[package]] +name = "anstyle-wincon" +version = "3.0.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "291e6a250ff86cd4a820112fb8898808a366d8f9f58ce16d1f538353ad55747d" +dependencies = [ + "anstyle", + "once_cell_polyfill", + "windows-sys", +] + +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + +[[package]] +name = "bitflags" +version = "2.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + +[[package]] +name = "cfg-if" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" + +[[package]] +name = "clap" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aa8876b300ab35ba921adea3dfd70157a46249b33f95c9084ae5709785478946" +dependencies = [ + "clap_builder", + "clap_derive", +] + +[[package]] +name = "clap_builder" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec0797fb7aeb1406c84efac526901f7ec3ead2124f946b494e72879d4b54704d" +dependencies = [ + "anstream", + "anstyle", + "clap_lex", + "strsim", +] + +[[package]] +name = "clap_derive" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9c751b79415d4e559e3d1fcf128e09e720eb673a06d26cf6f392d37d75b66e0" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "clap_lex" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1c133bc6a41be0d194c306b5506d15e6feeea7b1d6604bd3f8310dfb2ca96486" + +[[package]] +name = "colorchoice" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" + [[package]] name = "count-calls" version = "0.0.0" @@ -19,18 +142,121 @@ dependencies = [ "mirth-build", ] +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "crossbeam-deque" +version = "0.8.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "622f3fc73690be383c7214310406f28a90e6edeadc3cea882f9d71e495b9711a" +dependencies = [ + "crossbeam-epoch", + "crossbeam-utils", +] + +[[package]] +name = "crossbeam-epoch" +version = "0.9.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d" +dependencies = [ + "crossbeam-utils", +] + +[[package]] +name = "crossbeam-utils" +version = "0.8.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6" + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer", + "crypto-common", +] + +[[package]] +name = "either" +version = "1.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0e9c71c2167ca323c882b99918929403426e2373ea17242ff5653e0d5e1058be" + [[package]] name = "equivalent" version = "1.0.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" +[[package]] +name = "errno" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" +dependencies = [ + "libc", + "windows-sys", +] + +[[package]] +name = "fastrand" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" + +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + +[[package]] +name = "getrandom" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" +dependencies = [ + "cfg-if", + "libc", + "r-efi", +] + [[package]] name = "hashbrown" version = "0.17.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" +[[package]] +name = "heck" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" + [[package]] name = "indexmap" version = "2.14.2" @@ -41,6 +267,30 @@ dependencies = [ "hashbrown", ] +[[package]] +name = "is_terminal_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "libc" +version = "0.2.190" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce5d3ddc6d3fa000eb1536d85e147bfe31aacaba692ed6a876f95cb7c855be78" + +[[package]] +name = "linux-raw-sys" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" + [[package]] name = "memchr" version = "2.8.3" @@ -65,6 +315,23 @@ dependencies = [ "regex", ] +[[package]] +name = "mirth-lab" +version = "0.0.0" +dependencies = [ + "anyhow", + "clap", + "mirth-rewrite", + "rayon", + "regex", + "serde", + "serde_json", + "sha2", + "tempfile", + "wait-timeout", + "walkdir", +] + [[package]] name = "mirth-rewrite" version = "0.0.0" @@ -88,6 +355,18 @@ dependencies = [ "toml", ] +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "once_cell_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" + [[package]] name = "proc-macro2" version = "1.0.107" @@ -106,6 +385,32 @@ dependencies = [ "proc-macro2", ] +[[package]] +name = "r-efi" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" + +[[package]] +name = "rayon" +version = "1.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fb39b166781f92d482534ef4b4b1b2568f42613b53e5b6c160e24cfbfa30926d" +dependencies = [ + "either", + "rayon-core", +] + +[[package]] +name = "rayon-core" +version = "1.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "22e18b0f0062d30d4230b2e85ff77fdfe4326feb054b9783a3460d8435c8ab91" +dependencies = [ + "crossbeam-deque", + "crossbeam-utils", +] + [[package]] name = "regex" version = "1.13.1" @@ -135,6 +440,28 @@ version = "0.8.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" +[[package]] +name = "rustix" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "891efababe418670775f199f0d233d84843c227a0949a883ce15b37c78d6629d" +dependencies = [ + "bitflags", + "errno", + "libc", + "linux-raw-sys", + "windows-sys", +] + +[[package]] +name = "same-file" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502" +dependencies = [ + "winapi-util", +] + [[package]] name = "serde" version = "1.0.229" @@ -165,6 +492,19 @@ dependencies = [ "syn 3.0.6", ] +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + [[package]] name = "serde_spanned" version = "1.1.1" @@ -174,6 +514,23 @@ dependencies = [ "serde_core", ] +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures", + "digest", +] + +[[package]] +name = "strsim" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" + [[package]] name = "syn" version = "2.0.119" @@ -196,6 +553,19 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "tempfile" +version = "3.27.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" +dependencies = [ + "fastrand", + "getrandom", + "once_cell", + "rustix", + "windows-sys", +] + [[package]] name = "toml" version = "0.9.12+spec-1.1.0" @@ -241,12 +611,73 @@ version = "2.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "86a801b3cea342a06d468c8710662aa29e5e05e4f5c0d62f00bbb7f2ad7941c2" +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + [[package]] name = "unicode-ident" version = "1.0.26" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" +[[package]] +name = "utf8parse" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "wait-timeout" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09ac3b126d3914f9849036f826e054cbabdc8519970b8998ddaf3b5bd3c65f11" +dependencies = [ + "libc", +] + +[[package]] +name = "walkdir" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29790946404f91d9c5d06f9874efddea1dc06c5efe94541a7d6863108e3a5e4b" +dependencies = [ + "same-file", + "winapi-util", +] + +[[package]] +name = "winapi-util" +version = "0.1.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22" +dependencies = [ + "windows-sys", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + [[package]] name = "winnow" version = "0.7.15" @@ -258,3 +689,9 @@ name = "winnow" version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/Cargo.toml b/Cargo.toml index ca464a3..79d3b50 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -7,6 +7,7 @@ members = [ "crates/mirth-runtime", "crates/mirth-watch", "crates/mirth-rewrite", + "crates/mirth-lab", "examples/count-calls", ] exclude = ["fixtures"] diff --git a/crates/mirth-lab/Cargo.toml b/crates/mirth-lab/Cargo.toml new file mode 100644 index 0000000..149e5ca --- /dev/null +++ b/crates/mirth-lab/Cargo.toml @@ -0,0 +1,24 @@ +[package] +name = "mirth-lab" +description = "Checks (oracles) over rustc: differentials, metamorphic rewrites, invariants, sweeps over its test suites" +version.workspace = true +edition.workspace = true +license.workspace = true +publish.workspace = true + +[[bin]] +name = "mirth-lab" +path = "src/main.rs" + +[dependencies] +anyhow = "1" +clap = { version = "4.6", features = ["derive"] } +rayon = "1.12" +regex = "1" +serde = { version = "1", features = ["derive"] } +serde_json = "1" +sha2 = "0.10" +tempfile = "3" +walkdir = "2" +wait-timeout = "0.2" +mirth-rewrite = { path = "../mirth-rewrite" } diff --git a/crates/mirth-lab/src/artifacts.rs b/crates/mirth-lab/src/artifacts.rs new file mode 100644 index 0000000..e64e9e0 --- /dev/null +++ b/crates/mirth-lab/src/artifacts.rs @@ -0,0 +1,105 @@ +//! Compiler outputs in a form two builds can be compared in: archives taken apart, session +//! suffixes in member names and link metadata removed, everything hashed. + +use std::collections::BTreeMap; +use std::path::Path; +use std::sync::LazyLock; + +use regex::bytes::Regex; +use sha2::{Digest, Sha256}; + +/// `.<7 chars>.rcgu.o`: the per-session suffix of codegen-unit objects. +static SESSION: LazyLock = LazyLock::new(|| Regex::new(r"\.[0-9a-z]{7}(\.rcgu\.(?:o|dwo))").unwrap()); + +pub fn sha256(data: &[u8]) -> String { + let mut h = Sha256::new(); + h.update(data); + h.finalize().iter().map(|b| format!("{b:02x}")).collect() +} + +/// The members of a Unix ar archive (GNU format), names with session suffixes removed. +pub fn ar_members(data: &[u8]) -> BTreeMap> { + let mut members = BTreeMap::new(); + if !data.starts_with(b"!\n") { + members.insert("".into(), data.to_vec()); + return members; + } + let (mut names, mut pos): (&[u8], usize) = (b"", 8); + while pos + 60 <= data.len() { + let header = &data[pos..pos + 60]; + let raw_name = trim_end(&header[..16]); + let size: usize = std::str::from_utf8(&header[48..58]).ok().and_then(|s| s.trim().parse().ok()).unwrap_or(0); + let end = (pos + 60 + size).min(data.len()); + let body = &data[pos + 60..end]; + pos += 60 + size + (size & 1); + if raw_name == b"//" { + names = body; + continue; + } + if raw_name == b"/" || raw_name == b"/SYM64/" { + continue; + } + let mut name: Vec = raw_name.to_vec(); + if name.len() > 1 && name[0] == b'/' && name[1..].iter().all(u8::is_ascii_digit) { + let offset: usize = std::str::from_utf8(&name[1..]).unwrap().parse().unwrap_or(0); + let rest = names.get(offset..).unwrap_or(b""); + let len = rest.windows(2).position(|w| w == b"/\n").unwrap_or(rest.len()); + name = rest[..len].to_vec(); + } + while name.last() == Some(&b'/') { + name.pop(); + } + let name = SESSION.replace_all(&name, &b"$1"[..]); + members.insert(String::from_utf8_lossy(&name).into_owned(), body.to_vec()); + } + members +} + +fn trim_end(b: &[u8]) -> &[u8] { + let n = b.iter().rposition(|c| !c.is_ascii_whitespace()).map_or(0, |i| i + 1); + &b[..n] +} + +/// An rlib as {member: digest}, session suffixes removed from names and contents. +pub fn normalized_rlib(path: &Path) -> BTreeMap { + let data = std::fs::read(path).unwrap_or_default(); + ar_members(&data) + .into_iter() + .map(|(name, body)| (name, sha256(&SESSION.replace_all(&body, &b"$1"[..])))) + .collect() +} + +/// Every file in `dir` as {name: digest}; rlibs are normalized member by member. +pub fn digest_dir(dir: &Path) -> BTreeMap { + let mut out = BTreeMap::new(); + let Ok(entries) = std::fs::read_dir(dir) else { return out }; + for e in entries.flatten() { + let p = e.path(); + if !p.is_file() { + continue; + } + let name = e.file_name().to_string_lossy().into_owned(); + let digest = if p.extension().is_some_and(|x| x == "rlib") { + serde_json::to_string(&normalized_rlib(&p)).unwrap_or_default() + } else { + sha256(&std::fs::read(&p).unwrap_or_default()) + }; + out.insert(name, digest); + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn session_suffixes_removed() { + let mut ar = b"!\n".to_vec(); + let body = b"refers to t.abc1234.rcgu.o"; + ar.extend(format!("{:<16}{:<32}{:<10}`\n", "t.abc1234.rcgu.o/", "", body.len()).as_bytes()); + ar.extend(body); + let m = ar_members(&ar); + assert!(m.contains_key("t.rcgu.o"), "{m:?}"); + } +} diff --git a/crates/mirth-lab/src/driver.rs b/crates/mirth-lab/src/driver.rs new file mode 100644 index 0000000..d8a7c2e --- /dev/null +++ b/crates/mirth-lab/src/driver.rs @@ -0,0 +1,161 @@ +//! Running a check over many tests: a thread pool, one JSON line of results per test, findings +//! written to their own directories, and the frontier loop's options (pause at the first +//! finding, recheck only the tests with findings, leave known findings out). + +use std::collections::BTreeSet; +use std::fs::{self, File, OpenOptions}; +use std::io::{BufWriter, Write}; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::Mutex; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + +use rayon::prelude::*; +use serde::Serialize; +use serde_json::Value; + +use crate::uitest::Test; + +/// The options every sweep over UI tests takes. +#[derive(clap::Args, Clone, Debug)] +pub struct Sweep { + /// The rustc's test directory (`/tests/ui`). + #[arg(long)] + pub tests: PathBuf, + /// Where results, findings and scratch files go. + #[arg(long)] + pub work: PathBuf, + /// Only tests whose path contains this. + #[arg(long)] + pub only: Option, + /// A file of test paths to leave out (known findings). + #[arg(long)] + pub known: Option, + #[arg(long, default_value_t = 8)] + pub jobs: usize, + /// Stop starting new tests at the first finding, and exit 3 (the frontier loop). + #[arg(long)] + pub pause_on_finding: bool, + /// Run only the tests that have findings under /findings. + #[arg(long)] + pub recheck: bool, +} + +impl Sweep { + pub fn scratch(&self) -> PathBuf { + let s = self.work.join("scratch"); + let _ = fs::create_dir_all(&s); + s + } + + /// Filter `tests` by --only, --known and --recheck. + pub fn select(&self, tests: Vec) -> Vec { + let known: BTreeSet = self + .known + .as_ref() + .and_then(|k| fs::read_to_string(k).ok()) + .map(|t| t.split_whitespace().map(str::to_owned).collect()) + .unwrap_or_default(); + let wanted = self.recheck.then(|| findings_tests(&self.work)); + tests + .into_iter() + .filter(|t| { + !known.contains(&t.rel) + && self.only.as_ref().is_none_or(|o| t.rel.contains(o.as_str())) + && wanted.as_ref().is_none_or(|w| w.contains(&t.rel)) + }) + .collect() + } +} + +/// The tests named by /findings/*/finding.json. +pub fn findings_tests(work: &Path) -> BTreeSet { + let mut out = BTreeSet::new(); + if let Ok(dir) = fs::read_dir(work.join("findings")) { + for entry in dir.flatten() { + if let Ok(text) = fs::read_to_string(entry.path().join("finding.json")) + && let Ok(v) = serde_json::from_str::(&text) + && let Some(t) = v.get("test").and_then(Value::as_str) + { + out.insert(t.to_owned()); + } + } + } + out +} + +/// What a check reports for one test. +pub trait Record: Serialize + Send { + /// The findings, one line each (empty when there are none). + fn findings(&self) -> Vec; + fn test(&self) -> &str; +} + +/// Write a finding's directory: the test source, extra files, and `finding.json`. +pub fn write_finding(work: &Path, test: &Test, files: &[(String, Vec)], detail: &impl Serialize) { + let dir = work.join("findings").join(test.rel.replace('/', "__")); + let _ = fs::remove_dir_all(&dir); + if fs::create_dir_all(&dir).is_err() { + return; + } + let _ = fs::copy(&test.path, dir.join(test.file_name())); + for (name, bytes) in files { + let _ = fs::write(dir.join(name), bytes); + } + let mut value = serde_json::to_value(detail).unwrap_or(Value::Null); + if let Value::Object(map) = &mut value { + map.insert("test".into(), Value::String(test.rel.clone())); + map.insert("flags".into(), serde_json::json!(test.flags)); + map.insert("edition".into(), serde_json::json!(test.edition)); + } + let _ = fs::write(dir.join("finding.json"), serde_json::to_string_pretty(&value).unwrap_or_default()); +} + +/// Run `check` over `items` on `jobs` threads; results go to /results.jsonl (appended). +/// Returns exit 3 if paused at a finding, 0 otherwise. +pub fn drive(items: &[T], sweep: &Sweep, check: impl Fn(&T) -> R + Sync) -> ExitCode { + let _ = fs::create_dir_all(&sweep.work); + let file = OpenOptions::new().create(true).append(true).open(sweep.work.join("results.jsonl")); + let out: Mutex>> = Mutex::new(file.ok().map(BufWriter::new)); + let stop = AtomicBool::new(false); + let done = AtomicUsize::new(0); + let with_findings = AtomicUsize::new(0); + let total = items.len(); + let pool = rayon::ThreadPoolBuilder::new().num_threads(sweep.jobs.max(1)).build().expect("thread pool"); + pool.install(|| { + items.par_iter().for_each(|item| { + if stop.load(Ordering::Relaxed) { + return; + } + let record = check(item); + let findings = record.findings(); + if let Ok(mut guard) = out.lock() + && let Some(w) = guard.as_mut() + { + let _ = serde_json::to_writer(&mut *w, &record); + let _ = w.write_all(b"\n"); + let _ = w.flush(); + } + if !findings.is_empty() { + with_findings.fetch_add(1, Ordering::Relaxed); + println!("FINDING {}: {:?}", record.test(), findings); + if sweep.pause_on_finding { + stop.store(true, Ordering::Relaxed); + } + } + let n = done.fetch_add(1, Ordering::Relaxed) + 1; + if n % 100 == 0 { + println!("{n}/{total} done, {} with findings", with_findings.load(Ordering::Relaxed)); + } + }) + }); + let n = done.load(Ordering::Relaxed); + let f = with_findings.load(Ordering::Relaxed); + println!("{n} tests, {f} with findings"); + if f > 0 && sweep.pause_on_finding { ExitCode::from(3) } else { ExitCode::SUCCESS } +} + +/// A per-test scratch directory, removed when dropped. +pub fn scratch_dir(sweep: &Sweep) -> tempfile::TempDir { + tempfile::Builder::new().prefix("t").tempdir_in(sweep.scratch()).expect("scratch directory") +} diff --git a/crates/mirth-lab/src/lib.rs b/crates/mirth-lab/src/lib.rs new file mode 100644 index 0000000..0d8d916 --- /dev/null +++ b/crates/mirth-lab/src/lib.rs @@ -0,0 +1,9 @@ +//! Checks (oracles) over rustc: the shared parts. Each check is a subcommand of the +//! `mirth-lab` binary, in `src/tools/`; docs/checks.md says what each looks for and found. + +pub mod artifacts; +pub mod driver; +pub mod miri; +pub mod normalize; +pub mod rustc; +pub mod uitest; diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs new file mode 100644 index 0000000..09766a5 --- /dev/null +++ b/crates/mirth-lab/src/main.rs @@ -0,0 +1,49 @@ +//! `mirth-lab …`: the checks over rustc. Each subcommand's module documents what it +//! looks for; docs/checks.md has what each found. + +use std::process::ExitCode; + +use clap::{Parser, Subcommand}; + +mod tools { + pub mod crash_diff; + pub mod diag_check; + pub mod opt_diff; + pub mod solver_diff; +} + +#[derive(Parser)] +#[command(name = "mirth-lab", about = "Checks (oracles) over rustc")] +struct Cli { + #[command(subcommand)] + check: Check, +} + +#[derive(Subcommand)] +enum Check { + /// Behavior must not depend on optimization (opt levels, MIR opt levels, LTO, Cranelift). + OptDiff(tools::opt_diff::Args), + /// The old and new trait solvers, and NLL and Polonius, must agree. + SolverDiff(tools::solver_diff::Args), + /// rustc's internal checks (debug assertions, MIR validation) on every UI test. + CrashDiff(tools::crash_diff::Args), + /// Invariants of every diagnostic (no internal debug output, spans in bounds). + DiagCheck(tools::diag_check::Args), +} + +fn main() -> ExitCode { + let cli = Cli::parse(); + let result = match cli.check { + Check::OptDiff(a) => tools::opt_diff::run(a), + Check::SolverDiff(a) => tools::solver_diff::run(a), + Check::CrashDiff(a) => tools::crash_diff::run(a), + Check::DiagCheck(a) => tools::diag_check::run(a), + }; + match result { + Ok(code) => code, + Err(e) => { + eprintln!("mirth-lab: {e:#}"); + ExitCode::from(2) + } + } +} diff --git a/crates/mirth-lab/src/miri.rs b/crates/mirth-lab/src/miri.rs new file mode 100644 index 0000000..4811531 --- /dev/null +++ b/crates/mirth-lab/src/miri.rs @@ -0,0 +1,82 @@ +//! Interpreting a test with Miri (the pinned nightly's, with the sysroot `cargo miri setup` +//! makes). + +use std::path::{Path, PathBuf}; +use std::process::Command; +use std::time::Duration; + +use serde::Serialize; + +use crate::rustc::{Exit, run_command}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum MiriStatus { + /// Ran to the end, whatever the exit code. + Ok, + Ub, + Unsupported, + Error, + Ice, + Timeout, +} + +#[derive(Clone, Debug, Serialize)] +pub struct MiriRun { + pub status: MiriStatus, + pub exit: Exit, + pub stdout: String, + pub stderr: String, +} + +pub struct Miri { + pub binary: PathBuf, + pub sysroot: PathBuf, +} + +impl Miri { + /// The pinned toolchain's Miri and the default `cargo miri setup` sysroot. + pub fn pinned(toolchain: &str) -> Miri { + let home = PathBuf::from(std::env::var("HOME").unwrap_or_default()); + Miri { + binary: home.join(format!(".rustup/toolchains/{toolchain}-x86_64-unknown-linux-gnu/bin/miri")), + sysroot: home.join(".cache/miri"), + } + } + + pub fn run(&self, source: &Path, flags: &[String], edition: &str, extra: &[String], timeout: u64, cwd: &Path) -> MiriRun { + let mut cmd = Command::new(&self.binary); + cmd.arg("--sysroot") + .arg(&self.sysroot) + .arg(source) + .args(["--edition", edition]) + .args(["-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"]) + .args(["-Zmiri-disable-isolation", "-Zmiri-deterministic-floats"]) + .args(flags) + .args(extra) + .current_dir(cwd) + .env("RUSTC_BOOTSTRAP", "1") + .env("RUST_BACKTRACE", "0"); + let done = match run_command(cmd, Duration::from_secs(timeout)) { + Ok(d) => d, + Err(e) => { + return MiriRun { status: MiriStatus::Error, exit: Exit::Code(-1), stdout: String::new(), stderr: e.to_string() }; + } + }; + let err = done.stderr_text(); + let status = if done.exit == Exit::Timeout { + MiriStatus::Timeout + } else if err.contains("Undefined Behavior:") { + MiriStatus::Ub + } else if err.contains("unsupported operation") || err.contains("can't call foreign function") { + MiriStatus::Unsupported + } else if crate::rustc::is_ice(&err) { + MiriStatus::Ice + } else if done.exit == Exit::Code(1) && !err.contains("panicked") && err.lines().any(|l| l.starts_with("error")) { + MiriStatus::Error + } else { + MiriStatus::Ok + }; + MiriRun { status, exit: done.exit.clone(), stdout: done.stdout_text(), stderr: err } + } +} diff --git a/crates/mirth-lab/src/normalize.rs b/crates/mirth-lab/src/normalize.rs new file mode 100644 index 0000000..7180168 --- /dev/null +++ b/crates/mirth-lab/src/normalize.rs @@ -0,0 +1,49 @@ +//! What two runs of a program may print differently without the program behaving differently. + +use std::sync::LazyLock; + +use regex::Regex; + +static THREAD_ID: LazyLock = LazyLock::new(|| Regex::new(r"(thread '[^']*') \(\d+\)").unwrap()); +static STD_PATH: LazyLock = + LazyLock::new(|| Regex::new(r"\S*/lib/rustlib/src/rust/library/|/rustc/[0-9a-f]+/library/").unwrap()); +static REGISTRY: LazyLock = + LazyLock::new(|| Regex::new(r"\S*/registry/(src/)?[^/\s]+/([^/\s]+-\d[^/\s]*)/").unwrap()); +static TIMING: LazyLock = LazyLock::new(|| Regex::new(r"finished in \d+\.\d+s").unwrap()); + +/// Panic messages name the thread with its OS id; toolchains print std's and dependencies' paths +/// differently (in full with rust-src, remapped, relative); backtrace hints come and go. +pub fn stderr(text: &str) -> String { + let lines: Vec<&str> = text + .lines() + .filter(|l| !l.starts_with("note: run with `RUST_BACKTRACE") && !l.starts_with("note: Some details are omitted")) + .collect(); + let t = THREAD_ID.replace_all(&lines.join("\n"), "$1").into_owned(); + let t = STD_PATH.replace_all(&t, "library/").into_owned(); + REGISTRY.replace_all(&t, "/$2/").into_owned() +} + +/// The test harness prints how long tests took, and its result lines in completion order. +pub fn stdout(text: &str) -> String { + let t = TIMING.replace_all(text, "finished in …s").into_owned(); + let is_result = |l: &str| l.starts_with("test ") && l.contains(" ... "); + let mut results: Vec<&str> = t.split('\n').filter(|l| is_result(l)).collect(); + results.sort_unstable(); + let mut it = results.into_iter(); + t.split('\n').map(|l| if is_result(l) { it.next().unwrap_or(l) } else { l }).collect::>().join("\n") +} + +#[cfg(test)] +mod tests { + #[test] + fn thread_ids_and_paths() { + let s = super::stderr("thread 'main' (909942) panicked at /h/.rustup/x/lib/rustlib/src/rust/library/core/src/a.rs:1:2:"); + assert_eq!(s, "thread 'main' panicked at library/core/src/a.rs:1:2:"); + } + + #[test] + fn harness_order() { + let s = super::stdout("test b ... ok\ntest a ... ok\nfinished in 0.12s"); + assert_eq!(s, "test a ... ok\ntest b ... ok\nfinished in …s"); + } +} diff --git a/crates/mirth-lab/src/rustc.rs b/crates/mirth-lab/src/rustc.rs new file mode 100644 index 0000000..86c75ad --- /dev/null +++ b/crates/mirth-lab/src/rustc.rs @@ -0,0 +1,286 @@ +//! Running rustc (and the programs it builds, and Miri) with timeouts, and reading its JSON +//! diagnostics into types. + +use std::io::Read; +use std::path::{Path, PathBuf}; +use std::process::{Command, Stdio}; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; +use wait_timeout::ChildExt; + +/// How a command ended. +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum Exit { + Code(i32), + Signal(i32), + Timeout, +} + +/// What a finished command printed. +#[derive(Clone, Debug)] +pub struct Finished { + pub exit: Exit, + pub stdout: Vec, + pub stderr: Vec, +} + +impl Finished { + pub fn success(&self) -> bool { + self.exit == Exit::Code(0) + } + pub fn stderr_text(&self) -> String { + String::from_utf8_lossy(&self.stderr).into_owned() + } + pub fn stdout_text(&self) -> String { + String::from_utf8_lossy(&self.stdout).into_owned() + } +} + +/// Run `cmd` with stdin closed, its output captured, killed after `timeout`. +pub fn run_command(mut cmd: Command, timeout: Duration) -> std::io::Result { + cmd.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); + let mut child = cmd.spawn()?; + // Read both pipes on threads, so a chatty child cannot block on a full pipe. + let mut out = child.stdout.take().expect("piped"); + let mut err = child.stderr.take().expect("piped"); + let out_thread = std::thread::spawn(move || { + let mut v = Vec::new(); + let _ = out.read_to_end(&mut v); + v + }); + let err_thread = std::thread::spawn(move || { + let mut v = Vec::new(); + let _ = err.read_to_end(&mut v); + v + }); + let exit = match child.wait_timeout(timeout)? { + Some(status) => match status.code() { + Some(code) => Exit::Code(code), + None => { + use std::os::unix::process::ExitStatusExt; + Exit::Signal(status.signal().unwrap_or(0)) + } + }, + None => { + let _ = child.kill(); + let _ = child.wait(); + Exit::Timeout + } + }; + Ok(Finished { + exit, + stdout: out_thread.join().unwrap_or_default(), + stderr: err_thread.join().unwrap_or_default(), + }) +} + +pub fn is_ice(stderr: &str) -> bool { + stderr.contains("internal compiler error") + || stderr.contains("the compiler unexpectedly panicked") + || stderr.contains("rustc interrupted by SIG") +} + +/// The outcome of a compilation. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum Status { + Ok, + Error, + Ice, + Timeout, +} + +/// A compilation's result: its status, what it printed, and the program if it built one. +#[derive(Clone, Debug)] +pub struct Compiled { + pub status: Status, + pub stderr: String, + pub binary: Option, +} + +/// One rustc invocation, the way the checks compile a test. +pub struct Compile<'a> { + pub rustc: &'a Path, + pub source: &'a Path, + pub out_dir: &'a Path, + pub flags: &'a [String], + pub edition: &'a str, + pub extra: Vec, + /// `link`, `metadata`, `link,metadata`, ... + pub emit: &'a str, + pub json: bool, + pub timeout: Duration, + /// Run with `RUSTC_BOOTSTRAP=1` (the default; a stable user's view needs it off). + pub bootstrap: bool, +} + +impl<'a> Compile<'a> { + pub fn new(rustc: &'a Path, source: &'a Path, out_dir: &'a Path, flags: &'a [String], edition: &'a str) -> Self { + Compile { + rustc, + source, + out_dir, + flags, + edition, + extra: Vec::new(), + emit: "link", + json: false, + timeout: Duration::from_secs(300), + bootstrap: true, + } + } + + pub fn extra, S: Into>(mut self, extra: I) -> Self { + self.extra.extend(extra.into_iter().map(Into::into)); + self + } + + pub fn emit(mut self, emit: &'a str) -> Self { + self.emit = emit; + self + } + + pub fn json(mut self) -> Self { + self.json = true; + self + } + + pub fn timeout(mut self, secs: u64) -> Self { + self.timeout = Duration::from_secs(secs); + self + } + + pub fn run(self) -> Compiled { + let _ = std::fs::create_dir_all(self.out_dir); + let binary = self.out_dir.join("prog"); + let mut cmd = Command::new(self.rustc); + cmd.arg(self.source) + .args(["--edition", self.edition]) + .arg(format!("--emit={}", self.emit)) + .arg("-o") + .arg(&binary) + .args(["-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"]) + .arg(if self.json { "--error-format=json" } else { "--error-format=short" }) + .args(self.flags) + .args(&self.extra) + .current_dir(self.out_dir) + .env("RUST_BACKTRACE", "0"); + if self.bootstrap { + cmd.env("RUSTC_BOOTSTRAP", "1"); + } else { + cmd.env_remove("RUSTC_BOOTSTRAP"); + } + let done = match run_command(cmd, self.timeout) { + Ok(d) => d, + Err(e) => { + return Compiled { status: Status::Error, stderr: format!("spawning rustc: {e}"), binary: None }; + } + }; + let stderr = done.stderr_text(); + let status = match done.exit { + Exit::Timeout => Status::Timeout, + _ if is_ice(&stderr) => Status::Ice, + Exit::Code(0) => Status::Ok, + _ => Status::Error, + }; + let binary = (status == Status::Ok && self.emit.contains("link") && binary.exists()).then_some(binary); + Compiled { status, stderr, binary } + } +} + +/// What running a built program observed. +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub struct Observed { + pub exit: Exit, + pub stdout: String, + pub stderr: String, +} + +/// Run a program built by a check, from its own directory. +pub fn observe(binary: &Path, timeout_secs: u64, env: &[(&str, &str)]) -> Observed { + let mut cmd = Command::new(binary); + cmd.current_dir(binary.parent().unwrap_or(Path::new("."))).env("RUST_BACKTRACE", "0"); + for (k, v) in env { + cmd.env(k, v); + } + match run_command(cmd, Duration::from_secs(timeout_secs)) { + Ok(d) => Observed { exit: d.exit.clone(), stdout: d.stdout_text(), stderr: d.stderr_text() }, + Err(e) => Observed { exit: Exit::Code(-1), stdout: String::new(), stderr: format!("spawning: {e}") }, + } +} + +/// One diagnostic of `--error-format=json`. +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct Diagnostic { + pub message: String, + #[serde(default)] + pub code: Option, + pub level: String, + #[serde(default)] + pub spans: Vec, + #[serde(default)] + pub children: Vec, +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct Code { + pub code: String, +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct Span { + pub file_name: String, + pub byte_start: usize, + pub byte_end: usize, + pub line_start: usize, + pub line_end: usize, + #[serde(default)] + pub is_primary: bool, + #[serde(default)] + pub label: Option, + #[serde(default)] + pub suggested_replacement: Option, + #[serde(default)] + pub suggestion_applicability: Option, +} + +impl Diagnostic { + pub fn code(&self) -> &str { + self.code.as_ref().map_or("", |c| c.code.as_str()) + } + + /// This diagnostic and its children, depth first. + pub fn walk(&self) -> Vec<&Diagnostic> { + let mut out = vec![self]; + for c in &self.children { + out.extend(c.walk()); + } + out + } + + pub fn primary(&self) -> Option<&Span> { + self.spans.iter().find(|s| s.is_primary) + } +} + +/// The diagnostics in rustc's JSON stderr (lines that are not JSON are skipped). +pub fn diagnostics(stderr: &str) -> Vec { + stderr.lines().filter_map(|l| serde_json::from_str(l).ok()).collect() +} + +/// The error codes in human-readable stderr, sorted and deduplicated. +pub fn error_codes(stderr: &str) -> Vec { + static RE: std::sync::LazyLock = + std::sync::LazyLock::new(|| regex::Regex::new(r"error\[(E\d{4})\]").unwrap()); + let mut codes: Vec = RE.captures_iter(stderr).map(|c| c[1].to_owned()).collect(); + codes.sort(); + codes.dedup(); + codes +} + +/// The first line of stderr that starts an error, for reports. +pub fn first_error(stderr: &str) -> String { + stderr.lines().find(|l| l.contains("error")).unwrap_or("").chars().take(300).collect() +} diff --git a/crates/mirth-lab/src/tools/crash_diff.rs b/crates/mirth-lab/src/tools/crash_diff.rs new file mode 100644 index 0000000..e679d80 --- /dev/null +++ b/crates/mirth-lab/src/tools/crash_diff.rs @@ -0,0 +1,100 @@ +//! Internal checks on: what rustc's own invariants say about every UI test. +//! +//! Compiles each standalone UI test with the compiler under test and with a second one built +//! from the same tree with debug assertions (and overflow checks), adding `-Zvalidate-mir`. +//! A crash, failed assertion or MIR validation error under the second that the first does not +//! have is a finding. + +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{Compile, Status}; +use mirth_lab::uitest::{self, Kind, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// The release compiler under test. + #[arg(long)] + rustc: PathBuf, + /// The same compiler built with debug assertions. + #[arg(long)] + checked: PathBuf, + #[arg(long, default_value = "-Zvalidate-mir")] + extra: String, + #[command(flatten)] + sweep: Sweep, +} + +static MESSAGE: LazyLock> = LazyLock::new(|| { + [r"panicked at [^\n]*\n[^\n]*", r"internal compiler error: [^\n]*", r"broken MIR[^\n]*"] + .iter() + .map(|p| Regex::new(p).unwrap()) + .collect() +}); +static COMPILER_PATH: LazyLock = LazyLock::new(|| Regex::new(r"/\S+/compiler/").unwrap()); + +/// The first line saying what went wrong inside the compiler. +fn message(stderr: &str) -> String { + MESSAGE + .iter() + .find_map(|re| re.find(stderr)) + .map(|m| COMPILER_PATH.replace_all(m.as_str(), "compiler/").chars().take(400).collect()) + .unwrap_or_default() +} + +#[derive(Serialize)] +struct Rec { + test: String, + release: Status, + checked: Status, + #[serde(skip_serializing_if = "Option::is_none")] + note: Option, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn check(args: &Args, test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + let emit = if Kind::is_check(test.kind) { "metadata" } else { "link" }; + let a = Compile::new(&args.rustc, &test.path, &dir.path().join("release"), &test.flags, test.edition()) + .emit(emit) + .run(); + let b = Compile::new(&args.checked, &test.path, &dir.path().join("checked"), &test.flags, test.edition()) + .extra(args.extra.split_whitespace()) + .emit(emit) + .timeout(600) + .run(); + let mut rec = Rec { test: test.rel.clone(), release: a.status, checked: b.status, note: None, found: Vec::new() }; + if b.status == Status::Ice && a.status != Status::Ice { + let m = message(&b.stderr); + rec.found.push(format!("only with internal checks: {}", m.chars().take(160).collect::())); + let stderr: String = b.stderr.chars().rev().take(4000).collect::().chars().rev().collect(); + driver::write_finding( + &args.sweep.work, + test, + &[], + &serde_json::json!({ "extra": args.extra, "found": [{ "what": "only with internal checks", "message": m, "stderr": stderr }] }), + ); + } else if b.status == Status::Ice && a.status == Status::Ice && message(&a.stderr) != message(&b.stderr) { + rec.note = Some("both crash, differently".into()); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let tests = args.sweep.select(uitest::tests(&args.sweep.tests, uitest::ALL, |_| false)); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, t))) +} diff --git a/crates/mirth-lab/src/tools/diag_check.rs b/crates/mirth-lab/src/tools/diag_check.rs new file mode 100644 index 0000000..cafbbc7 --- /dev/null +++ b/crates/mirth-lab/src/tools/diag_check.rs @@ -0,0 +1,144 @@ +//! Diagnostic invariants: what every diagnostic rustc prints must satisfy, whatever the program. +//! +//! Compiles each standalone UI test with `--error-format=json` and checks every diagnostic, +//! children and suggestions included: +//! +//! - internal: user-facing text (message, labels, suggested code) contains compiler-internal +//! debug output (`DefId(`, region and type-variable debug names, `Opaque(DefId`, `{closure#0}` +//! in suggested code) +//! - span: a span outside its file (offsets past the end, lines past the last, start after end) +//! +//! Errors without any span and exact duplicates are notes. Tests that ask for compiler internals +//! on purpose (verbose printing, dump attributes) are left out. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{self, Compile, Status}; +use mirth_lab::uitest::{self, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[command(flatten)] + sweep: Sweep, +} + +static INTERNAL: LazyLock = LazyLock::new(|| { + Regex::new(r"DefId\(|\bRe(LateParam|Bound|Var|Early|Static)\b|'\{erased\}|\?\d+[tif]\b|'\^\d+(_\d+)?\b|Opaque\(DefId|\bAlias\((Projection|Opaque|Inherent|Free)|\bBoundRegionKind|\bDefPath\b|\bLocalDefId\b|\bTyKind::").unwrap() +}); +static INTERNAL_CODE: LazyLock = + LazyLock::new(|| Regex::new(r"\{closure#\d+\}|\{opaque#\d+\}|\{async block@|\{impl#\d+\}|\{constant#\d+\}").unwrap()); +static DEBUG_TEST: LazyLock = LazyLock::new(|| { + Regex::new(r"#!?\[rustc_(dump|effective_visibility|regions|variance|outlives|layout|abi|def_path|symbol_name|object_lifetime_default|evaluate_where_clauses|then_this_would_need|if_this_changed|clean|partition)").unwrap() +}); +static DEBUG_FLAG: LazyLock = LazyLock::new(|| Regex::new(r"verbose|-Zdump|unpretty|print-").unwrap()); +static SUMMARY: LazyLock = + LazyLock::new(|| Regex::new(r"^(aborting due to|could not compile|\d+ (previous )?errors?)").unwrap()); + +#[derive(Serialize, Clone, PartialEq, Eq, PartialOrd, Ord)] +struct Finding { + what: String, + detail: String, + code: String, +} + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + found: Vec, + notes: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn file_info(cache: &mut BTreeMap>, test_dir: &Path, name: &str) -> Option<(usize, usize)> { + *cache.entry(name.to_owned()).or_insert_with(|| { + let p = Path::new(name); + let data = std::fs::read(if p.is_absolute() { p.to_path_buf() } else { test_dir.join(p) }).ok()?; + Some((data.len(), data.iter().filter(|&&b| b == b'\n').count() + 1)) + }) +} + +fn check(args: &Args, test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + let c = Compile::new(&args.rustc, &test.path, dir.path(), &test.flags, test.edition()) + .emit("metadata") + .json() + .timeout(120) + .run(); + let mut rec = Rec { test: test.rel.clone(), skip: None, found: Vec::new(), notes: Vec::new() }; + if matches!(c.status, Status::Ice | Status::Timeout) { + rec.skip = Some(format!("{:?}", c.status).to_lowercase()); + return rec; + } + let diags = rustc::diagnostics(&c.stderr); + let test_dir = test.path.parent().unwrap_or(Path::new(".")); + let mut cache = BTreeMap::new(); + let mut found: BTreeSet = BTreeSet::new(); + let mut seen: BTreeMap<(String, String, String, Option<(String, usize, usize)>), usize> = BTreeMap::new(); + for diag in &diags { + let prim = diag.primary().map(|s| (s.file_name.clone(), s.byte_start, s.byte_end)); + *seen.entry((diag.level.clone(), diag.code().to_owned(), diag.message.clone(), prim)).or_default() += 1; + if diag.level == "error" && diag.spans.is_empty() && diag.children.is_empty() && !SUMMARY.is_match(&diag.message) { + rec.notes.push(format!("nowhere: {}", diag.message.chars().take(100).collect::())); + } + for node in diag.walk() { + let texts = std::iter::once(node.message.as_str()).chain(node.spans.iter().filter_map(|s| s.label.as_deref())); + for t in texts { + if let Some(m) = INTERNAL.find(t) { + found.insert(Finding { what: "internal".into(), detail: format!("{} :: {}", m.as_str(), t.chars().take(300).collect::()), code: diag.code().into() }); + } + } + for s in &node.spans { + if let Some(repl) = &s.suggested_replacement + && let Some(m) = INTERNAL.find(repl).or_else(|| INTERNAL_CODE.find(repl)) + { + found.insert(Finding { what: "internal".into(), detail: format!("{} :: suggests {:?}", m.as_str(), repl.chars().take(200).collect::()), code: diag.code().into() }); + } + if s.file_name.starts_with('<') { + continue; + } + if let Some((size, lines)) = file_info(&mut cache, test_dir, &s.file_name) + && (s.byte_start > s.byte_end || s.byte_end > size || s.line_start > lines || s.line_end > lines || s.line_start > s.line_end) + { + found.insert(Finding { what: "span".into(), detail: format!("{}:{}..{} lines {}..{} (file: {size} bytes, {lines} lines): {}", s.file_name, s.byte_start, s.byte_end, s.line_start, s.line_end, node.message.chars().take(150).collect::()), code: diag.code().into() }); + } + } + } + } + for ((level, _, message, _), n) in &seen { + if *n > 1 && (level == "error" || level == "warning") { + rec.notes.push(format!("duplicate x{n}: {}", message.chars().take(100).collect::())); + } + } + rec.found = found.iter().map(|f| format!("{}: {}", f.what, f.detail.chars().take(120).collect::())).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "found": found, "notes": rec.notes })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let tests = uitest::tests(&args.sweep.tests, uitest::ALL, |t| { + uitest::flag_matches(t, &DEBUG_FLAG) || DEBUG_TEST.is_match(&t.text) || t.text.contains("assumptions_on_binders") + }); + let tests = args.sweep.select(tests); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, t))) +} diff --git a/crates/mirth-lab/src/tools/opt_diff.rs b/crates/mirth-lab/src/tools/opt_diff.rs new file mode 100644 index 0000000..a94d8c7 --- /dev/null +++ b/crates/mirth-lab/src/tools/opt_diff.rs @@ -0,0 +1,185 @@ +//! Optimization differential: a program's behavior must not depend on how it was optimized. +//! +//! Builds each runnable UI test under a set of configurations (optimization levels, MIR +//! optimization levels, LTO, target CPU, the Cranelift backend) and runs it, with overflow checks +//! and debug assertions fixed, and compares exit status, stdout and stderr with the unoptimized +//! baseline (`-Copt-level=0 -Zmir-opt-level=0`). A baseline whose output varies between two runs +//! is skipped; a difference counts only if the configuration's binary repeats it. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::normalize; +use mirth_lab::rustc::{self, Compile, Observed, Status}; +use mirth_lab::uitest::{self, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// A rustc with the Cranelift backend (the pinned nightly's). + #[arg(long)] + cranelift: Option, + /// Comma-separated subset of configurations (`base` is always built). + #[arg(long)] + configs: Option, + #[command(flatten)] + sweep: Sweep, +} + +const FIXED: &[&str] = &["-Coverflow-checks=on", "-Cdebug-assertions=on", "-Cpanic=unwind", "-Cdebuginfo=0"]; +const CONFIGS: &[(&str, &[&str])] = &[ + ("base", &["-Copt-level=0", "-Zmir-opt-level=0"]), + ("O0", &["-Copt-level=0"]), + ("O0-mir4", &["-Copt-level=0", "-Zmir-opt-level=4"]), + ("O1", &["-Copt-level=1"]), + ("O2", &["-Copt-level=2"]), + ("O3", &["-Copt-level=3"]), + ("Os", &["-Copt-level=s"]), + ("Oz", &["-Copt-level=z"]), + ("O3-mir4", &["-Copt-level=3", "-Zmir-opt-level=4"]), + ("O3-lto", &["-Copt-level=3", "-Clto=fat", "-Ccodegen-units=1"]), + ("O2-cgu16", &["-Copt-level=2", "-Ccodegen-units=16"]), + ("O3-native", &["-Copt-level=3", "-Ctarget-cpu=native"]), + ("cranelift", &["-Copt-level=0", "-Zcodegen-backend=cranelift"]), +]; +/// Tests that choose these themselves are left out: the configuration would contradict them. +static OWN: LazyLock = LazyLock::new(|| { + Regex::new(r"^-O$|opt-level|mir-opt-level|overflow-checks|debug-assertions|codegen-backend|mir-enable-passes|^-Clto|lto=|target-cpu|panic=|-Cpanic|prefer-dynamic|-Zbuild-std").unwrap() +}); +/// Tests whose outcome legitimately depends on optimization: unspecified behavior (whether two +/// equal promoted constants share an address), stack usage, a backend's documented gaps. +const NOISE: &[(&str, &[&str])] = &[ + ("mir/mir_raw_fat_ptr.rs", &["cranelift"]), + ("codegen/StackColoring-not-blowup-stack-issue-40883.rs", &["O0-mir4", "O0"]), + ("attributes/fn-align-dyn.rs", &["cranelift"]), + ("backtrace/backtrace.rs", &["cranelift"]), +]; + +#[derive(Serialize)] +struct Finding { + config: String, + what: String, + #[serde(skip_serializing_if = "Option::is_none")] + base: Option, + #[serde(skip_serializing_if = "Option::is_none")] + got: Option, + #[serde(skip_serializing_if = "String::is_empty")] + stderr: String, +} + +#[derive(Serialize)] +struct Rec { + test: String, + build: BTreeMap, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn observe(binary: &Path) -> Observed { + // Every configuration's program runs from the same path: some tests print argv[0]. + let fixed = binary.parent().and_then(Path::parent).unwrap_or(Path::new(".")).join("run").join("prog"); + let _ = std::fs::create_dir_all(fixed.parent().unwrap()); + let _ = std::fs::copy(binary, &fixed); + let o = rustc::observe(&fixed, 20, &[]); + Observed { exit: o.exit, stdout: normalize::stdout(&o.stdout), stderr: normalize::stderr(&o.stderr) } +} + +fn check(args: &Args, configs: &[(&str, &[&str])], test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + let mut rec = Rec { test: test.rel.clone(), build: BTreeMap::new(), skip: None, found: Vec::new() }; + let mut runs: BTreeMap<&str, Observed> = BTreeMap::new(); + let mut stderrs: BTreeMap<&str, String> = BTreeMap::new(); + let unwinding = test.text.contains("catch_unwind") || test.text.contains("needs-unwind"); + for &(name, cfg) in configs { + // Cranelift does not unwind on this target yet (catch_unwind catches nothing). + if name == "cranelift" && unwinding { + continue; + } + let rustc = if name == "cranelift" { args.cranelift.as_deref().unwrap() } else { args.rustc.as_path() }; + let out = dir.path().join(name); + let c = Compile::new(rustc, &test.path, &out, &test.flags, test.edition()) + .extra(FIXED.iter().chain(cfg.iter()).copied()) + .run(); + rec.build.insert(name.to_owned(), c.status); + stderrs.insert(name, c.stderr.chars().rev().take(2000).collect::().chars().rev().collect()); + if let Some(b) = c.binary { + runs.insert(name, observe(&b)); + } + } + if rec.build.get("base") != Some(&Status::Ok) || !runs.contains_key("base") { + rec.skip = Some("baseline does not build".into()); + return rec; + } + let base = runs["base"].clone(); + let again = observe(&dir.path().join("base").join("prog")); + if again != base { + rec.skip = Some(format!("baseline is nondeterministic: {:?} vs {:?}", base, again).chars().take(600).collect()); + return rec; + } + let noise: &[&str] = NOISE.iter().find(|(t, _)| *t == test.rel).map_or(&[], |(_, c)| c); + let mut found = Vec::new(); + for &(name, _) in configs { + if name == "base" || !rec.build.contains_key(name) || noise.contains(&name) { + continue; + } + let status = rec.build[name]; + if status != Status::Ok { + // Cranelift's documented gaps (tail calls, some linkages and SIMD intrinsics) show as + // errors or as panics inside the backend. + if name == "cranelift" && (status == Status::Error || stderrs[name].contains("rustc_codegen_cranelift")) { + continue; + } + found.push(Finding { config: name.into(), what: format!("build {status:?}").to_lowercase(), base: None, got: None, stderr: stderrs[name].clone() }); + continue; + } + let Some(got) = runs.get(name) else { continue }; + if name == "cranelift" && got.stderr.contains("failed to initiate panic") { + continue; + } + if *got != base { + let retry = observe(&dir.path().join(name).join("prog")); + if retry != base && retry == *got { + let diff: Vec<&str> = [("exit", got.exit != base.exit), ("stdout", got.stdout != base.stdout), ("stderr", got.stderr != base.stderr)] + .into_iter() + .filter_map(|(k, d)| d.then_some(k)) + .collect(); + found.push(Finding { config: name.into(), what: format!("run differs: {}", diff.join(",")), base: Some(base.clone()), got: Some(got.clone()), stderr: String::new() }); + } + } + } + rec.found = found.iter().map(|f| format!("{}: {}", f.config, f.what)).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "fixed": FIXED, "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let wanted: Option> = args.configs.as_deref().map(|c| c.split(',').collect()); + let configs: Vec<(&str, &[&str])> = CONFIGS + .iter() + .copied() + .filter(|(n, _)| *n == "base" || wanted.as_ref().is_none_or(|w| w.contains(n))) + .filter(|(n, _)| *n != "cranelift" || args.cranelift.is_some()) + .collect(); + let tests = uitest::tests(&args.sweep.tests, uitest::RUNNABLE, |t| uitest::flag_matches(t, &OWN)); + let tests = args.sweep.select(tests); + println!("{} tests, configurations: {}", tests.len(), configs.iter().map(|c| c.0).collect::>().join(", ")); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &configs, t))) +} diff --git a/crates/mirth-lab/src/tools/solver_diff.rs b/crates/mirth-lab/src/tools/solver_diff.rs new file mode 100644 index 0000000..6d8c5ed --- /dev/null +++ b/crates/mirth-lab/src/tools/solver_diff.rs @@ -0,0 +1,138 @@ +//! Solver differential: the old and new trait solvers, and NLL and Polonius, must agree. +//! +//! Compiles each standalone UI test four ways (old solver = `-Znext-solver=coherence`, the +//! default new solver, each with `-Zpolonius=next`) and compares with the old solver and NLL: a +//! crash, a timeout or a different verdict is a finding; both rejecting with different error +//! codes is a note. A program accepted only by a non-reference configuration is interpreted with +//! Miri under that configuration: undefined behavior means the other one accepted something +//! unsound. Tests that name a solver or Polonius in their headers are left out. + +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::miri::{Miri, MiriStatus}; +use mirth_lab::rustc::{self, Compile, Status}; +use mirth_lab::uitest::{self, Kind, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// The toolchain whose Miri interprets one-sided acceptances. + #[arg(long, default_value = "nightly-2026-10-06")] + miri_toolchain: String, + #[command(flatten)] + sweep: Sweep, +} + +const CONFIGS: &[(&str, &[&str])] = &[ + ("old", &["-Znext-solver=coherence"]), + ("next", &[]), + ("old-polonius", &["-Znext-solver=coherence", "-Zpolonius=next"]), + ("next-polonius", &["-Zpolonius=next"]), +]; +static OWN: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)next-solver|polonius|^//@\s*revisions:.*\bnext\b").unwrap()); + +#[derive(Serialize)] +struct Result1 { + status: Status, + codes: Vec, + stderr: String, +} + +#[derive(Serialize)] +struct Finding { + config: String, + what: String, + #[serde(skip_serializing_if = "Option::is_none")] + miri: Option, + stderr: String, +} + +#[derive(Serialize)] +struct Rec { + test: String, + status: Vec<(String, Status)>, + found: Vec, + notes: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn tail(s: &str, n: usize) -> String { + let v: Vec = s.chars().collect(); + v[v.len().saturating_sub(n)..].iter().collect() +} + +fn check(args: &Args, miri: &Miri, test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + // Metadata for check tests; build and run tests get a full build (monomorphization errors). + let emit = if Kind::is_check(test.kind) { "metadata" } else { "link" }; + let results: Vec<(&str, Result1)> = CONFIGS + .iter() + .map(|&(name, cfg)| { + let c = Compile::new(&args.rustc, &test.path, &dir.path().join(name), &test.flags, test.edition()) + .extra(cfg.iter().copied()) + .emit(emit) + .timeout(120) + .run(); + (name, Result1 { status: c.status, codes: rustc::error_codes(&c.stderr), stderr: tail(&c.stderr, 2500) }) + }) + .collect(); + let reference = &results[0].1; + let (mut found, mut notes) = (Vec::new(), Vec::new()); + for (name, r) in &results[1..] { + if r.status == Status::Ice && reference.status != Status::Ice { + found.push(Finding { config: name.to_string(), what: "ice".into(), miri: None, stderr: r.stderr.clone() }); + } else if r.status == Status::Timeout && reference.status != Status::Timeout { + found.push(Finding { config: name.to_string(), what: "timeout".into(), miri: None, stderr: String::new() }); + } else if matches!((r.status, reference.status), (Status::Ok, Status::Error) | (Status::Error, Status::Ok)) { + let accepted_by = if r.status == Status::Ok { name } else { "old" }; + let mut what = format!("verdict: old {:?}, {name} {:?}", reference.status, r.status).to_lowercase(); + let mut miri_status = None; + if test.text.contains("fn main") { + let cfg = CONFIGS.iter().find(|c| c.0 == accepted_by).map_or(&[][..], |c| c.1); + let extra: Vec = cfg.iter().map(|s| s.to_string()).collect(); + let m = miri.run(&test.path, &test.flags, test.edition(), &extra, 120, dir.path()); + if m.status == MiriStatus::Ub { + what.push_str(&format!("; Miri: UB under {accepted_by}")); + } + miri_status = Some(format!("{:?}", m.status)); + } + let stderr = if r.status == Status::Error { r.stderr.clone() } else { reference.stderr.clone() }; + found.push(Finding { config: name.to_string(), what, miri: miri_status, stderr }); + } else if r.status == Status::Error && reference.status == Status::Error && r.codes != reference.codes { + notes.push(format!("{name} codes {:?} vs {:?}", r.codes, reference.codes)); + } + } + let rec = Rec { + test: test.rel.clone(), + status: results.iter().map(|(n, r)| (n.to_string(), r.status)).collect(), + found: found.iter().map(|f| format!("{}: {}", f.config, f.what)).collect(), + notes, + }; + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let miri = Miri::pinned(&args.miri_toolchain); + let tests = uitest::tests(&args.sweep.tests, uitest::ALL, |t| OWN.is_match(&t.text)); + let tests = args.sweep.select(tests); + println!("{} tests, configurations: old, next, old-polonius, next-polonius", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &miri, t))) +} diff --git a/crates/mirth-lab/src/uitest.rs b/crates/mirth-lab/src/uitest.rs new file mode 100644 index 0000000..60387d4 --- /dev/null +++ b/crates/mirth-lab/src/uitest.rs @@ -0,0 +1,182 @@ +//! rustc's UI tests as the checks use them: their `//@` headers (the first revision of a test +//! with revisions), and which of them can be compiled on their own on this host. + +use std::path::{Path, PathBuf}; +use std::sync::LazyLock; + +use regex::Regex; +use walkdir::WalkDir; + +/// What a test expects of the compiler, from its `//@ ` header. +#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum Kind { + CheckPass, + BuildPass, + RunPass, + CheckFail, + BuildFail, + RunFail, +} + +impl Kind { + fn parse(key: &str) -> Option { + Some(match key { + "check-pass" => Kind::CheckPass, + "build-pass" => Kind::BuildPass, + "run-pass" => Kind::RunPass, + "check-fail" => Kind::CheckFail, + "build-fail" => Kind::BuildFail, + "run-fail" => Kind::RunFail, + _ => return None, + }) + } + + /// Whether type checking is all the test asks for (metadata is enough to compile it). + pub fn is_check(kind: Option) -> bool { + matches!(kind, None | Some(Kind::CheckPass) | Some(Kind::CheckFail)) + } +} + +/// Every kind, and tests without a kind header. +pub const ALL: &[Option] = &[ + Some(Kind::CheckPass), + Some(Kind::BuildPass), + Some(Kind::RunPass), + Some(Kind::CheckFail), + Some(Kind::BuildFail), + Some(Kind::RunFail), + None, +]; +pub const RUNNABLE: &[Option] = &[Some(Kind::RunPass), Some(Kind::RunFail)]; + +/// A UI test, as one of its revisions compiles it. +#[derive(Clone, Debug)] +pub struct Test { + pub path: PathBuf, + /// The path below the test root, as findings and known lists name it. + pub rel: String, + pub text: String, + pub flags: Vec, + pub edition: Option, + pub kind: Option, + pub revision: Option, +} + +impl Test { + pub fn edition(&self) -> &str { + self.edition.as_deref().unwrap_or("2015") + } + + pub fn file_name(&self) -> &str { + self.path.file_name().and_then(|n| n.to_str()).unwrap_or("test.rs") + } +} + +static REVISIONS: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^//@\s*revisions:\s*(.*)$").unwrap()); +static DIRECTIVE: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$").unwrap()); +/// Tests that need more than one file, another target, or a tool this host may lack. +static NOT_STANDALONE: LazyLock = LazyLock::new(|| { + Regex::new( + r"(?m)^//@\s*(aux-build|aux-crate|aux-bin|aux-codegen-backend|proc-macro|add-minicore|needs-llvm-components|needs-sanitizer|needs-profiler|needs-rust-lld|needs-enzyme|ignore-x86_64|ignore-linux|ignore-unix|ignore-64bit|known-bug|rustc-env|unset-rustc-env)\b", + ) + .unwrap() +}); +static ONLY: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^//@\s*only-(\S+)").unwrap()); +/// `only-` directives this host satisfies. +const HOST: &[&str] = &["x86_64", "linux", "unix", "64bit", "elf", "gnu"]; + +/// The headers of a test's first revision. +pub fn headers(text: &str) -> (Vec, Option, Option, Option) { + let revision = + REVISIONS.captures(text).and_then(|c| c[1].split_whitespace().next().map(str::to_owned)); + let (mut flags, mut edition, mut kind) = (Vec::new(), None, None); + for c in DIRECTIVE.captures_iter(text) { + if let Some(only) = c.get(1) + && !revision.as_deref().is_some_and(|r| only.as_str().split(',').any(|o| o == r)) + { + continue; + } + let value = c.get(3).map_or("", |v| v.as_str().trim()); + match &c[2] { + "compile-flags" => flags.extend(value.split_whitespace().map(str::to_owned)), + "edition" => edition = value.split_whitespace().next().map(str::to_owned), + key => { + if let Some(k) = Kind::parse(key) { + kind = Some(k); + } + } + } + } + if let Some(r) = &revision { + flags.extend(["--cfg".to_owned(), r.clone()]); + } + (flags, edition, kind, revision) +} + +/// Whether a test can be compiled alone on this host. +pub fn standalone(text: &str) -> bool { + !NOT_STANDALONE.is_match(text) + && ONLY.captures_iter(text).all(|c| HOST.iter().any(|h| c[1].starts_with(h))) +} + +/// The standalone tests under `root` of the given kinds, minus those `skip` rejects. +pub fn tests(root: &Path, kinds: &[Option], skip: impl Fn(&Test) -> bool) -> Vec { + let mut out = Vec::new(); + let mut paths: Vec = WalkDir::new(root) + .into_iter() + .filter_map(Result::ok) + .filter(|e| { + e.file_type().is_file() + && e.path().extension().is_some_and(|x| x == "rs") + && !e.path().components().any(|c| c.as_os_str() == "auxiliary") + }) + .map(|e| e.into_path()) + .collect(); + paths.sort(); + for path in paths { + let Ok(text) = std::fs::read_to_string(&path) else { continue }; + if !standalone(&text) { + continue; + } + let (flags, edition, kind, revision) = headers(&text); + if !kinds.contains(&kind) { + continue; + } + let rel = path.strip_prefix(root).unwrap_or(&path).to_string_lossy().into_owned(); + let test = Test { path, rel, text, flags, edition, kind, revision }; + if !skip(&test) { + out.push(test); + } + } + out +} + +/// Whether the test (or its flags) names one of these, as a substring of a flag. +pub fn flag_matches(test: &Test, re: &Regex) -> bool { + test.flags.iter().any(|f| re.is_match(f)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn first_revision_headers() { + let text = "//@ revisions: a b\n//@[a] compile-flags: -Zfoo\n//@[b] compile-flags: -Zbar\n//@ edition: 2021\n//@[a] check-pass\n"; + let (flags, edition, kind, rev) = headers(text); + assert_eq!(flags, ["-Zfoo", "--cfg", "a"]); + assert_eq!(edition.as_deref(), Some("2021")); + assert_eq!(kind, Some(Kind::CheckPass)); + assert_eq!(rev.as_deref(), Some("a")); + } + + #[test] + fn host_only_directives() { + assert!(standalone("//@ only-x86_64\n")); + assert!(standalone("//@ only-linux\n")); + assert!(!standalone("//@ only-aarch64\n")); + assert!(!standalone("//@ aux-build: x.rs\n")); + } +} diff --git a/crates/mirth-rewrite/src/lib.rs b/crates/mirth-rewrite/src/lib.rs new file mode 100644 index 0000000..a1aa938 --- /dev/null +++ b/crates/mirth-rewrite/src/lib.rs @@ -0,0 +1,319 @@ +//! Meaning-preserving rewrites of a Rust file, for metamorphic testing of rustc: a rewritten +//! program must get the same verdict (and the same error codes) as the original. +//! +//! [`rewrite`] returns the rewritten file, or why there is none. Comments are not kept (the +//! tokens are printed back), so line numbers change: compare verdicts and error codes, not +//! spans. The `mirth-rewrite` binary wraps it for the command line. +//! +//! Rewrites: +//! +//! - `identity`: the file printed back unchanged: the baseline for the others, since printing +//! tokens back drops comments and moves every line. +//! - `generic-wrap`: the body of each free function with plain parameters moves into a generic +//! inner function, called with `()` for its unused type parameter. What the body does is the +//! same; it is now checked in a generic context and instantiated. +//! - `alias`: each struct, enum and union gets a type alias with the same generic parameters, +//! and every type mentioning it by its bare name mentions the alias instead. +//! - `reorder`: the top-level items in reverse order (item order does not matter in Rust; +//! files with `macro_rules!` or item-position macro calls are left out, where it does). +//! - `unused`: an unused function, struct and trait added at the end. + +use std::collections::{BTreeMap, BTreeSet}; + +use proc_macro2::Span; +use quote::{ToTokens, format_ident, quote}; +use syn::visit_mut::VisitMut; +use syn::{FnArg, GenericParam, Ident, Item, ItemFn, Pat, Type, TypePath}; + +/// The rewrites, `identity` first (the baseline for the others). +pub const REWRITES: [&str; 5] = ["identity", "generic-wrap", "alias", "reorder", "unused"]; + +/// What a rewrite of a file came to. +#[derive(Debug, PartialEq, Eq)] +pub enum Outcome { + /// The rewritten file. + Rewritten(String), + /// `syn` does not parse the file. + DoesNotParse, + /// The rewrite has nothing to change in this file. + DoesNotApply, + /// No rewrite has this name. + Unknown, +} + +/// Rewrite `text` (a whole Rust file) by the rewrite called `name`. +pub fn rewrite(name: &str, text: &str) -> Outcome { + let Ok(mut file) = syn::parse_file(text) else { return Outcome::DoesNotParse }; + let applied = match name { + "generic-wrap" => generic_wrap(&mut file), + "alias" => alias(&mut file), + "reorder" => reorder(&mut file), + "unused" => unused(&mut file), + "identity" => true, + _ => return Outcome::Unknown, + }; + if applied { Outcome::Rewritten(file.into_token_stream().to_string()) } else { Outcome::DoesNotApply } +} + +/// Moves the body of `fn f(a: A, b: B) -> R { body }` into +/// `fn __mirth_inner<__MirthT>(a: A, b: B) -> R { body }`, called as `__mirth_inner::<()>(a, b)`. +fn generic_wrap(file: &mut syn::File) -> bool { + let mut applied = false; + for item in &mut file.items { + if let Item::Fn(f) = item + && wrappable(f) + { + wrap(f); + applied = true; + } + } + applied +} + +fn wrappable(f: &ItemFn) -> bool { + let sig = &f.sig; + // Plain functions only: no generics (an inner fn cannot use the outer's parameters), no + // `impl Trait`, no qualifiers that change how the body runs, no `self`, no attributes that + // name the function (`#[test]`, `#[no_mangle]`, `#[track_caller]` would change meaning). + sig.generics.params.is_empty() + && sig.generics.where_clause.is_none() + && sig.constness.is_none() + && sig.asyncness.is_none() + && sig.unsafety.is_none() + && sig.abi.is_none() + && sig.variadic.is_none() + && f.attrs.iter().all(|a| a.path().is_ident("allow") || a.path().is_ident("inline")) + && !sig.to_token_stream().to_string().contains("impl ") + && !sig.to_token_stream().to_string().contains('\'') + // Parameters without attributes: a `#[cfg]`-ed out one cannot be passed on by name. + && sig.inputs.iter().all(|arg| matches!(arg, FnArg::Typed(t) if t.attrs.is_empty() && matches!(&*t.pat, Pat::Ident(p) if p.by_ref.is_none() && p.subpat.is_none()))) +} + +fn wrap(f: &mut ItemFn) { + let sig = &f.sig; + let inputs = &sig.inputs; + let output = &sig.output; + let names: Vec<&Ident> = sig + .inputs + .iter() + .map(|arg| match arg { + FnArg::Typed(t) => match &*t.pat { + Pat::Ident(p) => &p.ident, + _ => unreachable!("checked in wrappable"), + }, + FnArg::Receiver(_) => unreachable!("checked in wrappable"), + }) + .collect(); + let body = &f.block; + // No attributes (a test may `forbid` the lint they would allow); names no lint objects to. + let new: syn::Block = syn::parse_quote!({ + fn _mirth_inner(#inputs) #output #body + _mirth_inner::<()>(#(#names),*) + }); + // Parameters declared `mut` are mutated in the body, now the inner function's. + for arg in f.sig.inputs.iter_mut() { + if let FnArg::Typed(t) = arg + && let Pat::Ident(p) = &mut *t.pat + { + p.mutability = None; + } + } + *f.block = new; +} + +/// `type __MirthAlias_S = S;` for each struct, enum and union `S`, and every +/// type that names `S` by its bare name names the alias instead. +fn alias(file: &mut syn::File) -> bool { + // Names also used for generic parameters somewhere: a bare path may mean the parameter. + struct Params(BTreeSet); + impl VisitMut for Params { + fn visit_generic_param_mut(&mut self, p: &mut GenericParam) { + match p { + GenericParam::Type(t) => self.0.insert(t.ident.to_string()), + GenericParam::Const(c) => self.0.insert(c.ident.to_string()), + GenericParam::Lifetime(_) => false, + }; + syn::visit_mut::visit_generic_param_mut(self, p); + } + } + let mut params_seen = Params(BTreeSet::new()); + params_seen.visit_file_mut(&mut file.clone()); + // Types defined more than once (a nested item shadowing a top-level one): a bare name may + // mean either. + struct Defined(BTreeMap); + impl VisitMut for Defined { + fn visit_item_struct_mut(&mut self, i: &mut syn::ItemStruct) { + *self.0.entry(i.ident.to_string()).or_default() += 1; + syn::visit_mut::visit_item_struct_mut(self, i); + } + fn visit_item_enum_mut(&mut self, i: &mut syn::ItemEnum) { + *self.0.entry(i.ident.to_string()).or_default() += 1; + syn::visit_mut::visit_item_enum_mut(self, i); + } + fn visit_item_union_mut(&mut self, i: &mut syn::ItemUnion) { + *self.0.entry(i.ident.to_string()).or_default() += 1; + syn::visit_mut::visit_item_union_mut(self, i); + } + fn visit_item_type_mut(&mut self, i: &mut syn::ItemType) { + *self.0.entry(i.ident.to_string()).or_default() += 1; + syn::visit_mut::visit_item_type_mut(self, i); + } + } + let mut defined = Defined(BTreeMap::new()); + defined.visit_file_mut(&mut file.clone()); + let mut aliases = BTreeMap::new(); + let mut new_items = Vec::new(); + for item in &file.items { + let (ident, generics) = match item { + Item::Struct(s) => (&s.ident, &s.generics), + Item::Enum(e) => (&e.ident, &e.generics), + Item::Union(u) => (&u.ident, &u.generics), + _ => continue, + }; + if params_seen.0.contains(&ident.to_string()) || defined.0.get(&ident.to_string()) != Some(&1) { + continue; + } + // Lifetime parameters elide differently through an alias; defaults may name other + // parameters: such types are left alone. + if generics.params.iter().any(|p| match p { + GenericParam::Lifetime(_) => true, + GenericParam::Type(t) => t.default.is_some(), + GenericParam::Const(c) => c.default.is_some(), + }) { + continue; + } + let alias = format_ident!("_MirthAlias{}", ident); + // The type's own `cfg`s: an alias of a configured-out type would name nothing. + let cfgs: Vec<&syn::Attribute> = match item { + Item::Struct(x) => &x.attrs, + Item::Enum(x) => &x.attrs, + Item::Union(x) => &x.attrs, + _ => unreachable!(), + } + .iter() + // `cfg` only: a `cfg_attr` may expand to a `derive`, which an alias cannot have. + .filter(|a| a.path().is_ident("cfg")) + .collect(); + // The alias's parameters: the type's, without bounds (aliases ignore them), with defaults. + let mut params = generics.clone(); + params.where_clause = None; + for p in params.params.iter_mut() { + match p { + GenericParam::Type(t) => { + // `?Sized` must stay: without it the alias requires `Sized`. + let maybe: Vec = t + .bounds + .iter() + .filter(|b| matches!(b, syn::TypeParamBound::Trait(tb) if matches!(tb.modifier, syn::TraitBoundModifier::Maybe(_)))) + .cloned() + .collect(); + t.bounds = maybe.into_iter().collect(); + if t.bounds.is_empty() { + t.colon_token = None; + } + } + GenericParam::Lifetime(l) => { + l.bounds.clear(); + l.colon_token = None; + } + GenericParam::Const(_) => {} + } + } + let args: Vec = generics + .params + .iter() + .map(|p| match p { + GenericParam::Type(t) => t.ident.to_token_stream(), + GenericParam::Lifetime(l) => l.lifetime.to_token_stream(), + GenericParam::Const(c) => c.ident.to_token_stream(), + }) + .collect(); + let target = if args.is_empty() { quote!(#ident) } else { quote!(#ident<#(#args),*>) }; + new_items.push(syn::parse_quote!( + #(#cfgs)* + type #alias #params = #target; + )); + aliases.insert(ident.to_string(), alias); + } + if aliases.is_empty() { + return false; + } + struct Rename<'a> { + aliases: &'a BTreeMap, + count: usize, + } + impl VisitMut for Rename<'_> { + fn visit_type_path_mut(&mut self, ty: &mut TypePath) { + if ty.qself.is_none() + && ty.path.leading_colon.is_none() + && ty.path.segments.len() == 1 + && let Some(alias) = self.aliases.get(&ty.path.segments[0].ident.to_string()) + { + ty.path.segments[0].ident = Ident::new(&alias.to_string(), Span::call_site()); + self.count += 1; + } + syn::visit_mut::visit_type_path_mut(self, ty); + } + // Inside a type's own definition, `Self`-like uses must stay (a recursive type through + // its alias would be a cycle in the alias), so definitions are not visited. + fn visit_item_struct_mut(&mut self, _: &mut syn::ItemStruct) {} + fn visit_item_enum_mut(&mut self, _: &mut syn::ItemEnum) {} + fn visit_item_union_mut(&mut self, _: &mut syn::ItemUnion) {} + // Macros' tokens are not types to `syn`; derive input stays as it is. + fn visit_macro_mut(&mut self, _: &mut syn::Macro) {} + // In an inline module the bare name reaches the type through a `use`, the alias would + // need one too: modules are left as they are. + fn visit_item_mod_mut(&mut self, _: &mut syn::ItemMod) {} + // A receiver typed with the impl's own name (`self: &mut Test`) elides lifetimes like + // `&mut self`; through an alias it does not (resolution does not see through aliases). + fn visit_receiver_mut(&mut self, _: &mut syn::Receiver) {} + // The same rule compares a receiver with the impl's self type: that stays as written. + fn visit_item_impl_mut(&mut self, imp: &mut syn::ItemImpl) { + let self_ty = std::mem::replace(&mut *imp.self_ty, Type::Verbatim(Default::default())); + syn::visit_mut::visit_item_impl_mut(self, imp); + *imp.self_ty = self_ty; + } + } + let mut rename = Rename { aliases: &aliases, count: 0 }; + for item in &mut file.items { + rename.visit_item_mut(item); + } + if rename.count == 0 { + return false; + } + file.items.extend(new_items); + true +} + +fn reorder(file: &mut syn::File) -> bool { + let has_macros = file.items.iter().any(|item| matches!(item, Item::Macro(_))); + if has_macros || file.items.len() < 2 { + return false; + } + // `use` and `extern crate` first, as written, so that preludes and `#[macro_use]` stay put. + let (mut head, mut rest): (Vec, Vec) = + file.items.drain(..).partition(|item| matches!(item, Item::Use(_) | Item::ExternCrate(_))); + rest.reverse(); + head.extend(rest); + file.items = head; + true +} + +fn unused(file: &mut syn::File) -> bool { + let names: BTreeSet = file + .items + .iter() + .filter_map(|item| match item { + Item::Fn(f) => Some(f.sig.ident.to_string()), + _ => None, + }) + .collect(); + if names.contains("_mirth_unused") { + return false; + } + // Leading underscores keep `dead_code` quiet without an attribute a test could `forbid`. + file.items.push(syn::parse_quote!(fn _mirth_unused() {})); + file.items.push(syn::parse_quote!(struct _MirthUnused;)); + file.items.push(syn::parse_quote!(trait _MirthUnusedTrait {})); + true +} diff --git a/crates/mirth-rewrite/src/main.rs b/crates/mirth-rewrite/src/main.rs index 93b037e..9a791fd 100644 --- a/crates/mirth-rewrite/src/main.rs +++ b/crates/mirth-rewrite/src/main.rs @@ -1,320 +1,28 @@ -//! Meaning-preserving rewrites of a Rust file, for metamorphic testing of rustc: a rewritten -//! program must get the same verdict (and the same error codes) as the original. -//! -//! mirth-rewrite -//! -//! prints the rewritten file to stdout; exits 2 when the file does not parse (as `syn` sees -//! Rust) and 3 when the rewrite does not apply to it. Comments are not kept (the tokens are -//! printed back), so line numbers change: compare verdicts and error codes, not spans. -//! -//! Rewrites: -//! -//! - `identity`: the file printed back unchanged: the baseline for the others, since printing -//! tokens back drops comments and moves every line. -//! - `generic-wrap`: the body of each free function with plain parameters moves into a generic -//! inner function, called with `()` for its unused type parameter. What the body does is the -//! same; it is now checked in a generic context and instantiated. -//! - `alias`: each struct, enum and union gets a type alias with the same generic parameters, -//! and every type mentioning it by its bare name mentions the alias instead. -//! - `reorder`: the top-level items in reverse order (item order does not matter in Rust; -//! files with `macro_rules!` or item-position macro calls are left out, where it does). -//! - `unused`: an unused function, struct and trait added at the end. +//! `mirth-rewrite `: prints the rewritten file to stdout; exits 2 when the file +//! does not parse (as `syn` sees Rust) and 3 when the rewrite does not apply to it. The rewrites +//! are documented in the library. -use std::collections::{BTreeMap, BTreeSet}; use std::process::exit; -use proc_macro2::Span; -use quote::{ToTokens, format_ident, quote}; -use syn::visit_mut::VisitMut; -use syn::{FnArg, GenericParam, Ident, Item, ItemFn, Pat, Type, TypePath}; +use mirth_rewrite::{Outcome, REWRITES, rewrite}; fn main() { let args: Vec = std::env::args().collect(); if args.len() != 3 { - eprintln!("usage: mirth-rewrite "); + eprintln!("usage: mirth-rewrite <{}> ", REWRITES.join("|")); exit(64); } let text = std::fs::read_to_string(&args[2]).unwrap_or_else(|error| { eprintln!("{}: {error}", args[2]); exit(64) }); - let Ok(mut file) = syn::parse_file(&text) else { exit(2) }; - let applied = match args[1].as_str() { - "generic-wrap" => generic_wrap(&mut file), - "alias" => alias(&mut file), - "reorder" => reorder(&mut file), - "unused" => unused(&mut file), - "identity" => true, - other => { - eprintln!("unknown rewrite {other}"); + match rewrite(&args[1], &text) { + Outcome::Rewritten(out) => println!("{out}"), + Outcome::DoesNotParse => exit(2), + Outcome::DoesNotApply => exit(3), + Outcome::Unknown => { + eprintln!("unknown rewrite {}", args[1]); exit(64) } - }; - if !applied { - exit(3); } - println!("{}", file.into_token_stream()); -} - -/// Moves the body of `fn f(a: A, b: B) -> R { body }` into -/// `fn __mirth_inner<__MirthT>(a: A, b: B) -> R { body }`, called as `__mirth_inner::<()>(a, b)`. -fn generic_wrap(file: &mut syn::File) -> bool { - let mut applied = false; - for item in &mut file.items { - if let Item::Fn(f) = item - && wrappable(f) - { - wrap(f); - applied = true; - } - } - applied -} - -fn wrappable(f: &ItemFn) -> bool { - let sig = &f.sig; - // Plain functions only: no generics (an inner fn cannot use the outer's parameters), no - // `impl Trait`, no qualifiers that change how the body runs, no `self`, no attributes that - // name the function (`#[test]`, `#[no_mangle]`, `#[track_caller]` would change meaning). - sig.generics.params.is_empty() - && sig.generics.where_clause.is_none() - && sig.constness.is_none() - && sig.asyncness.is_none() - && sig.unsafety.is_none() - && sig.abi.is_none() - && sig.variadic.is_none() - && f.attrs.iter().all(|a| a.path().is_ident("allow") || a.path().is_ident("inline")) - && !sig.to_token_stream().to_string().contains("impl ") - && !sig.to_token_stream().to_string().contains('\'') - // Parameters without attributes: a `#[cfg]`-ed out one cannot be passed on by name. - && sig.inputs.iter().all(|arg| matches!(arg, FnArg::Typed(t) if t.attrs.is_empty() && matches!(&*t.pat, Pat::Ident(p) if p.by_ref.is_none() && p.subpat.is_none()))) -} - -fn wrap(f: &mut ItemFn) { - let sig = &f.sig; - let inputs = &sig.inputs; - let output = &sig.output; - let names: Vec<&Ident> = sig - .inputs - .iter() - .map(|arg| match arg { - FnArg::Typed(t) => match &*t.pat { - Pat::Ident(p) => &p.ident, - _ => unreachable!("checked in wrappable"), - }, - FnArg::Receiver(_) => unreachable!("checked in wrappable"), - }) - .collect(); - let body = &f.block; - // No attributes (a test may `forbid` the lint they would allow); names no lint objects to. - let new: syn::Block = syn::parse_quote!({ - fn _mirth_inner(#inputs) #output #body - _mirth_inner::<()>(#(#names),*) - }); - // Parameters declared `mut` are mutated in the body, now the inner function's. - for arg in f.sig.inputs.iter_mut() { - if let FnArg::Typed(t) = arg - && let Pat::Ident(p) = &mut *t.pat - { - p.mutability = None; - } - } - *f.block = new; -} - -/// `type __MirthAlias_S = S;` for each struct, enum and union `S`, and every -/// type that names `S` by its bare name names the alias instead. -fn alias(file: &mut syn::File) -> bool { - // Names also used for generic parameters somewhere: a bare path may mean the parameter. - struct Params(BTreeSet); - impl VisitMut for Params { - fn visit_generic_param_mut(&mut self, p: &mut GenericParam) { - match p { - GenericParam::Type(t) => self.0.insert(t.ident.to_string()), - GenericParam::Const(c) => self.0.insert(c.ident.to_string()), - GenericParam::Lifetime(_) => false, - }; - syn::visit_mut::visit_generic_param_mut(self, p); - } - } - let mut params_seen = Params(BTreeSet::new()); - params_seen.visit_file_mut(&mut file.clone()); - // Types defined more than once (a nested item shadowing a top-level one): a bare name may - // mean either. - struct Defined(BTreeMap); - impl VisitMut for Defined { - fn visit_item_struct_mut(&mut self, i: &mut syn::ItemStruct) { - *self.0.entry(i.ident.to_string()).or_default() += 1; - syn::visit_mut::visit_item_struct_mut(self, i); - } - fn visit_item_enum_mut(&mut self, i: &mut syn::ItemEnum) { - *self.0.entry(i.ident.to_string()).or_default() += 1; - syn::visit_mut::visit_item_enum_mut(self, i); - } - fn visit_item_union_mut(&mut self, i: &mut syn::ItemUnion) { - *self.0.entry(i.ident.to_string()).or_default() += 1; - syn::visit_mut::visit_item_union_mut(self, i); - } - fn visit_item_type_mut(&mut self, i: &mut syn::ItemType) { - *self.0.entry(i.ident.to_string()).or_default() += 1; - syn::visit_mut::visit_item_type_mut(self, i); - } - } - let mut defined = Defined(BTreeMap::new()); - defined.visit_file_mut(&mut file.clone()); - let mut aliases = BTreeMap::new(); - let mut new_items = Vec::new(); - for item in &file.items { - let (ident, generics) = match item { - Item::Struct(s) => (&s.ident, &s.generics), - Item::Enum(e) => (&e.ident, &e.generics), - Item::Union(u) => (&u.ident, &u.generics), - _ => continue, - }; - if params_seen.0.contains(&ident.to_string()) || defined.0.get(&ident.to_string()) != Some(&1) { - continue; - } - // Lifetime parameters elide differently through an alias; defaults may name other - // parameters: such types are left alone. - if generics.params.iter().any(|p| match p { - GenericParam::Lifetime(_) => true, - GenericParam::Type(t) => t.default.is_some(), - GenericParam::Const(c) => c.default.is_some(), - }) { - continue; - } - let alias = format_ident!("_MirthAlias{}", ident); - // The type's own `cfg`s: an alias of a configured-out type would name nothing. - let cfgs: Vec<&syn::Attribute> = match item { - Item::Struct(x) => &x.attrs, - Item::Enum(x) => &x.attrs, - Item::Union(x) => &x.attrs, - _ => unreachable!(), - } - .iter() - // `cfg` only: a `cfg_attr` may expand to a `derive`, which an alias cannot have. - .filter(|a| a.path().is_ident("cfg")) - .collect(); - // The alias's parameters: the type's, without bounds (aliases ignore them), with defaults. - let mut params = generics.clone(); - params.where_clause = None; - for p in params.params.iter_mut() { - match p { - GenericParam::Type(t) => { - // `?Sized` must stay: without it the alias requires `Sized`. - let maybe: Vec = t - .bounds - .iter() - .filter(|b| matches!(b, syn::TypeParamBound::Trait(tb) if matches!(tb.modifier, syn::TraitBoundModifier::Maybe(_)))) - .cloned() - .collect(); - t.bounds = maybe.into_iter().collect(); - if t.bounds.is_empty() { - t.colon_token = None; - } - } - GenericParam::Lifetime(l) => { - l.bounds.clear(); - l.colon_token = None; - } - GenericParam::Const(_) => {} - } - } - let args: Vec = generics - .params - .iter() - .map(|p| match p { - GenericParam::Type(t) => t.ident.to_token_stream(), - GenericParam::Lifetime(l) => l.lifetime.to_token_stream(), - GenericParam::Const(c) => c.ident.to_token_stream(), - }) - .collect(); - let target = if args.is_empty() { quote!(#ident) } else { quote!(#ident<#(#args),*>) }; - new_items.push(syn::parse_quote!( - #(#cfgs)* - type #alias #params = #target; - )); - aliases.insert(ident.to_string(), alias); - } - if aliases.is_empty() { - return false; - } - struct Rename<'a> { - aliases: &'a BTreeMap, - count: usize, - } - impl VisitMut for Rename<'_> { - fn visit_type_path_mut(&mut self, ty: &mut TypePath) { - if ty.qself.is_none() - && ty.path.leading_colon.is_none() - && ty.path.segments.len() == 1 - && let Some(alias) = self.aliases.get(&ty.path.segments[0].ident.to_string()) - { - ty.path.segments[0].ident = Ident::new(&alias.to_string(), Span::call_site()); - self.count += 1; - } - syn::visit_mut::visit_type_path_mut(self, ty); - } - // Inside a type's own definition, `Self`-like uses must stay (a recursive type through - // its alias would be a cycle in the alias), so definitions are not visited. - fn visit_item_struct_mut(&mut self, _: &mut syn::ItemStruct) {} - fn visit_item_enum_mut(&mut self, _: &mut syn::ItemEnum) {} - fn visit_item_union_mut(&mut self, _: &mut syn::ItemUnion) {} - // Macros' tokens are not types to `syn`; derive input stays as it is. - fn visit_macro_mut(&mut self, _: &mut syn::Macro) {} - // In an inline module the bare name reaches the type through a `use`, the alias would - // need one too: modules are left as they are. - fn visit_item_mod_mut(&mut self, _: &mut syn::ItemMod) {} - // A receiver typed with the impl's own name (`self: &mut Test`) elides lifetimes like - // `&mut self`; through an alias it does not (resolution does not see through aliases). - fn visit_receiver_mut(&mut self, _: &mut syn::Receiver) {} - // The same rule compares a receiver with the impl's self type: that stays as written. - fn visit_item_impl_mut(&mut self, imp: &mut syn::ItemImpl) { - let self_ty = std::mem::replace(&mut *imp.self_ty, Type::Verbatim(Default::default())); - syn::visit_mut::visit_item_impl_mut(self, imp); - *imp.self_ty = self_ty; - } - } - let mut rename = Rename { aliases: &aliases, count: 0 }; - for item in &mut file.items { - rename.visit_item_mut(item); - } - if rename.count == 0 { - return false; - } - file.items.extend(new_items); - true -} - -fn reorder(file: &mut syn::File) -> bool { - let has_macros = file.items.iter().any(|item| matches!(item, Item::Macro(_))); - if has_macros || file.items.len() < 2 { - return false; - } - // `use` and `extern crate` first, as written, so that preludes and `#[macro_use]` stay put. - let (mut head, mut rest): (Vec, Vec) = - file.items.drain(..).partition(|item| matches!(item, Item::Use(_) | Item::ExternCrate(_))); - rest.reverse(); - head.extend(rest); - file.items = head; - true -} - -fn unused(file: &mut syn::File) -> bool { - let names: BTreeSet = file - .items - .iter() - .filter_map(|item| match item { - Item::Fn(f) => Some(f.sig.ident.to_string()), - _ => None, - }) - .collect(); - if names.contains("_mirth_unused") { - return false; - } - // Leading underscores keep `dead_code` quiet without an attribute a test could `forbid`. - file.items.push(syn::parse_quote!(fn _mirth_unused() {})); - file.items.push(syn::parse_quote!(struct _MirthUnused;)); - file.items.push(syn::parse_quote!(trait _MirthUnusedTrait {})); - true } From ea442973aa35dc3b5c86e4e935f2a96f36c86670 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 07:20:50 +0000 Subject: [PATCH 08/23] mirth-lab: miri-diff, rewrite-diff (mirth-rewrite in process), suggest-diff (identical to the Python sweep: 7,357 suggestions, 159 tests with findings) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- crates/mirth-lab/src/main.rs | 12 ++ crates/mirth-lab/src/tools/miri_diff.rs | 172 ++++++++++++++++++ crates/mirth-lab/src/tools/rewrite_diff.rs | 146 ++++++++++++++++ crates/mirth-lab/src/tools/suggest_diff.rs | 193 +++++++++++++++++++++ 4 files changed, 523 insertions(+) create mode 100644 crates/mirth-lab/src/tools/miri_diff.rs create mode 100644 crates/mirth-lab/src/tools/rewrite_diff.rs create mode 100644 crates/mirth-lab/src/tools/suggest_diff.rs diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs index 09766a5..ee8423b 100644 --- a/crates/mirth-lab/src/main.rs +++ b/crates/mirth-lab/src/main.rs @@ -8,8 +8,11 @@ use clap::{Parser, Subcommand}; mod tools { pub mod crash_diff; pub mod diag_check; + pub mod miri_diff; pub mod opt_diff; + pub mod rewrite_diff; pub mod solver_diff; + pub mod suggest_diff; } #[derive(Parser)] @@ -29,6 +32,12 @@ enum Check { CrashDiff(tools::crash_diff::Args), /// Invariants of every diagnostic (no internal debug output, spans in bounds). DiagCheck(tools::diag_check::Args), + /// Accepted safe programs are UB-free under Miri; MIR optimizations and native code agree with Miri. + MiriDiff(tools::miri_diff::Args), + /// Meaning-preserving rewrites (generic-wrap, alias, reorder, unused) keep the verdict. + RewriteDiff(tools::rewrite_diff::Args), + /// Machine-applicable suggestions, applied one at a time, keep the program compiling. + SuggestDiff(tools::suggest_diff::Args), } fn main() -> ExitCode { @@ -38,6 +47,9 @@ fn main() -> ExitCode { Check::SolverDiff(a) => tools::solver_diff::run(a), Check::CrashDiff(a) => tools::crash_diff::run(a), Check::DiagCheck(a) => tools::diag_check::run(a), + Check::MiriDiff(a) => tools::miri_diff::run(a), + Check::RewriteDiff(a) => tools::rewrite_diff::run(a), + Check::SuggestDiff(a) => tools::suggest_diff::run(a), }; match result { Ok(code) => code, diff --git a/crates/mirth-lab/src/tools/miri_diff.rs b/crates/mirth-lab/src/tools/miri_diff.rs new file mode 100644 index 0000000..896e4f2 --- /dev/null +++ b/crates/mirth-lab/src/tools/miri_diff.rs @@ -0,0 +1,172 @@ +//! Miri differential: an accepted safe program must be free of undefined behavior, MIR +//! optimizations must not introduce any, and the compiled program must do what Miri says. +//! +//! For each runnable UI test, interprets it with Miri at `-Zmir-opt-level` 0, 2 and 4 and builds +//! and runs it natively. Findings: UB at level 0 in a test without `unsafe` code; UB only after +//! MIR optimization; Miri's exit status or stdout changing with the MIR level; the native +//! program's exit status or stdout differing from Miri's. Tests Miri cannot run are skipped, as +//! are comparisons for threaded tests (scheduling differs) and tests asserting what Rust leaves +//! unspecified (function pointer equality, zero-sized addresses), which Miri varies on purpose. + +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::miri::{Miri, MiriRun, MiriStatus}; +use mirth_lab::normalize; +use mirth_lab::rustc::{self, Compile, Exit}; +use mirth_lab::uitest::{self, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[arg(long, default_value = "nightly-2026-10-06")] + miri_toolchain: String, + /// Seconds per Miri run. + #[arg(long, default_value_t = 120)] + timeout: u64, + #[command(flatten)] + sweep: Sweep, +} + +const FIXED: &[&str] = &["-Coverflow-checks=on", "-Cdebug-assertions=on"]; +/// Miri without preemption: threads switch only where they block, the same at every MIR level. +const MIRI_FLAGS: &[&str] = &["-Zmiri-preemption-rate=0"]; +const LEVELS: &[(&str, &[&str])] = &[("miri0", &[]), ("miri2", &["-Zmir-opt-level=2"]), ("miri4", &["-Zmir-opt-level=4"])]; +/// Tests asserting what Rust leaves unspecified, which Miri varies on purpose: function pointer +/// equality, the addresses of zero-sized values, stack addresses, function alignment; and one +/// Miri limitation (`.init_array` functions called without glibc's arguments). +const UNSPECIFIED: &[&str] = &[ + "consts/const-extern-function.rs", + "consts/zst_no_llvm_alloc.rs", + "layout/null-pointer-optimization.rs", + "mir/mir_misc_casts.rs", + "mir/mir_coercions.rs", + "extern/extern-compare-with-return-type.rs", + "fn/fn-ptr-trait-run.rs", + "mir/mir_raw_fat_ptr.rs", + "codegen/StackColoring-not-blowup-stack-issue-40883.rs", + "attributes/fn-align-dyn.rs", + "runtime/stdout-before-main.rs", +]; +static THREADS: LazyLock = LazyLock::new(|| Regex::new(r"thread::(spawn|scope)|std::sync::mpsc|\bspawn\(").unwrap()); +static OWN: LazyLock = LazyLock::new(|| { + Regex::new(r"^-O$|opt-level|overflow-checks|debug-assertions|codegen-backend|mir-enable-passes|panic=|-Cpanic|prefer-dynamic|-Zbuild-std|-Clink|-Ctarget").unwrap() +}); +/// Code Miri cannot interpret, and compile-time output that would land in Miri's stdout. +static NOT_FOR_MIRI: LazyLock = LazyLock::new(|| { + Regex::new(r#"\basm!|global_asm!|naked_asm!|extern\s+"C"\s*\{|#\[link\(|std::process::Command|\bfork\b|libc::|dlopen|std::os::unix::process|trace_macros|log_syntax"#).unwrap() +}); + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + #[serde(skip_serializing_if = "Option::is_none")] + note: Option, + miri: Vec<(String, MiriStatus)>, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn clean(text: &str, test: &Test) -> String { + // argv[0]: the source file under Miri, the binary natively. + let t = text.replace(&test.path.to_string_lossy().into_owned(), ""); + let t = regex_argv0().replace_all(&t, "").into_owned(); + normalize::stdout(&normalize::stderr(&t)) +} + +fn regex_argv0() -> &'static Regex { + static RE: LazyLock = LazyLock::new(|| Regex::new(r"\S*/native/prog\b").unwrap()); + &RE +} + +fn tail(s: &str, n: usize) -> String { + let v: Vec = s.chars().collect(); + v[v.len().saturating_sub(n)..].iter().collect() +} + +fn check(args: &Args, miri: &Miri, test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + let mut rec = Rec { test: test.rel.clone(), skip: None, note: None, miri: Vec::new(), found: Vec::new() }; + let native = Compile::new(&args.rustc, &test.path, &dir.path().join("native"), &test.flags, test.edition()) + .extra(FIXED.iter().copied().chain(["-Copt-level=0"])) + .run(); + let Some(binary) = native.binary else { + rec.skip = Some(format!("native build {:?}", native.status).to_lowercase()); + return rec; + }; + let mut outs: Vec<(Exit, String)> = + (0..3).map(|_| rustc::observe(&binary, 20, &[])).map(|o| (o.exit, clean(&o.stdout, test))).collect(); + outs.dedup(); + if outs.len() > 1 { + rec.skip = Some("native run is nondeterministic".into()); + return rec; + } + let (native_exit, native_out) = outs.pop().unwrap(); + let mut runs: Vec<(&str, MiriRun)> = Vec::new(); + for &(name, extra) in LEVELS { + let flags: Vec = FIXED.iter().chain(MIRI_FLAGS).chain(extra).map(|s| s.to_string()).collect(); + let mut m = miri.run(&test.path, &test.flags, test.edition(), &flags, args.timeout, dir.path()); + m.stdout = clean(&m.stdout, test); + if name == "miri0" && matches!(m.status, MiriStatus::Unsupported | MiriStatus::Error | MiriStatus::Timeout) { + rec.skip = Some(format!("miri {:?}", m.status).to_lowercase()); + return rec; + } + runs.push((name, m)); + } + rec.miri = runs.iter().map(|(n, m)| (n.to_string(), m.status)).collect(); + let unspecified = UNSPECIFIED.contains(&test.rel.as_str()); + let threaded = THREADS.is_match(&test.text); + let safe = !test.text.contains("unsafe"); + let m0 = &runs[0].1; + let mut found: Vec = Vec::new(); + // UB in a program without `unsafe` code can only be the compiler's. + if m0.status == MiriStatus::Ub && safe && !unspecified { + found.push(serde_json::json!({ "what": "ub (safe code)", "stderr": tail(&m0.stderr, 3000) })); + } else if m0.status == MiriStatus::Ub { + rec.note = Some("ub in a test with unsafe code".into()); + } + for (name, m) in &runs[1..] { + if m.status == MiriStatus::Ub && m0.status != MiriStatus::Ub { + found.push(serde_json::json!({ "what": format!("ub-opt ({name})"), "stderr": tail(&m.stderr, 3000) })); + } else if m.status == MiriStatus::Ice { + found.push(serde_json::json!({ "what": format!("ice ({name})"), "stderr": tail(&m.stderr, 3000) })); + } else if m.status == MiriStatus::Ok && m0.status == MiriStatus::Ok && !threaded && (&m.exit, &m.stdout) != (&m0.exit, &m0.stdout) { + found.push(serde_json::json!({ "what": format!("opt-differs ({name})"), "miri0": tail(&m0.stdout, 1500), "got": tail(&m.stdout, 1500) })); + } + } + if m0.status == MiriStatus::Ok && !threaded && !unspecified { + // Miri exits 1 on a panic that reaches main, native code 101. + let exit_m = if m0.exit == Exit::Code(1) && m0.stderr.contains("panicked") { Exit::Code(101) } else { m0.exit.clone() }; + if (exit_m, &m0.stdout) != (native_exit.clone(), &native_out) { + found.push(serde_json::json!({ "what": "native", "miri": [m0.exit, tail(&m0.stdout, 1500)], "native": [native_exit, tail(&native_out, 1500)] })); + } + } + rec.found = found.iter().map(|f| f["what"].as_str().unwrap_or("").to_owned()).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "fixed": FIXED, "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let miri = Miri::pinned(&args.miri_toolchain); + let tests = uitest::tests(&args.sweep.tests, uitest::RUNNABLE, |t| uitest::flag_matches(t, &OWN) || NOT_FOR_MIRI.is_match(&t.text)); + let tests = args.sweep.select(tests); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &miri, t))) +} diff --git a/crates/mirth-lab/src/tools/rewrite_diff.rs b/crates/mirth-lab/src/tools/rewrite_diff.rs new file mode 100644 index 0000000..84e63b1 --- /dev/null +++ b/crates/mirth-lab/src/tools/rewrite_diff.rs @@ -0,0 +1,146 @@ +//! Equivalent rewrites: rewriting a program into an equivalent one must not change its verdict. +//! +//! Each standalone UI test is printed back unchanged (the `identity` rewrite: the baseline, +//! since printing drops comments and moves lines) and rewritten by each of mirth-rewrite's +//! rewrites, in process. Each version is compiled with lints capped (a full build when there is +//! a `fn main`), and a changed verdict is a finding; both rejecting with different error codes +//! is a note. Tests whose baseline differs from the original are left out (the printer cannot +//! represent them), as are tests that name files by relative path or have no `core`. + +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{self, Compile, Status}; +use mirth_lab::uitest::{self, Kind, Test}; +use mirth_rewrite::{Outcome, REWRITES, rewrite}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// Comma-separated subset of rewrites. + #[arg(long)] + rewrites: Option, + #[command(flatten)] + sweep: Sweep, +} + +static NOT_MOVABLE: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)^\s*(pub(\([^)]*\))?\s+)?mod\s+\w+\s*;|include(_str|_bytes)?!|#\[path|#!\[no_core\]").unwrap()); +/// Item order matters to textual macro scoping: no reordering where macros are defined. +static ORDER_MATTERS: LazyLock = LazyLock::new(|| Regex::new(r"macro_rules!|#\[macro_use\]|macro\s+\w+").unwrap()); +/// Differences that are resource limits, or findings already recorded. +const NOISE: &[(&str, &str)] = &[ + ("consts/chained-constants-stackoverflow.rs", "reorder"), // 10,000 chained consts: query depth + ("consts/interior-mut-const-via-union.rs", "generic-wrap"), // finding 25 + // recursion_limit = "6": evaluation order nests the query stack one level deeper + ("traits/next-solver/overflow/dont-lower-depth-for-witness-and-rigid-opaque.rs", "reorder"), + ("imports/ambiguous-9.rs", "reorder"), // finding 28 + ("imports/ambiguous-14.rs", "reorder"), + ("imports/overwrite-different-ambig-2.rs", "reorder"), +]; + +#[derive(Serialize, Clone)] +struct Verdict { + status: Status, + codes: Vec, + stderr: String, +} + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + applied: Vec, + found: Vec, + notes: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn verdict(args: &Args, test: &Test, source: &std::path::Path, out: &std::path::Path) -> Verdict { + // A full build whenever there is a program: generic-wrap moves errors to monomorphization. + let emit = if test.text.contains("fn main") || !Kind::is_check(test.kind) { "link" } else { "metadata" }; + let c = Compile::new(&args.rustc, source, out, &test.flags, test.edition()) + .extra(["--cap-lints=warn"]) + .emit(emit) + .timeout(120) + .run(); + let stderr: String = c.stderr.chars().rev().take(2500).collect::().chars().rev().collect(); + Verdict { status: c.status, codes: rustc::error_codes(&c.stderr), stderr } +} + +fn check(args: &Args, names: &[&str], test: &Test) -> Rec { + let mut rec = Rec { test: test.rel.clone(), skip: None, applied: Vec::new(), found: Vec::new(), notes: Vec::new() }; + if NOT_MOVABLE.is_match(&test.text) { + rec.skip = Some("uses files by path or has no core".into()); + return rec; + } + let Outcome::Rewritten(identity) = rewrite("identity", &test.text) else { + rec.skip = Some("does not parse".into()); + return rec; + }; + let dir = driver::scratch_dir(&args.sweep); + let base_src = dir.path().join("identity.rs"); + let _ = std::fs::write(&base_src, &identity); + let original = verdict(args, test, &test.path, &dir.path().join("original")); + let base = verdict(args, test, &base_src, &dir.path().join("identity")); + if (original.status, &original.codes) != (base.status, &base.codes) { + rec.skip = Some("printing changes the verdict".into()); + return rec; + } + let mut kept: Vec<(String, Vec)> = vec![("identity.rs".into(), identity.into_bytes())]; + let (mut found, mut notes) = (Vec::new(), Vec::new()); + for &name in names { + if name == "identity" + || (name == "reorder" && ORDER_MATTERS.is_match(&test.text)) + || NOISE.contains(&(test.rel.as_str(), name)) + // generic_const_exprs requires bounds in generic contexts that a concrete one does not. + || (name == "generic-wrap" && test.text.contains("generic_const_exprs")) + { + continue; + } + let Outcome::Rewritten(text) = rewrite(name, &test.text) else { continue }; + rec.applied.push(name.into()); + let src = dir.path().join(format!("{name}.rs")); + let _ = std::fs::write(&src, &text); + let v = verdict(args, test, &src, &dir.path().join(name)); + let entry = |what: String| serde_json::json!({ "rewrite": name, "what": what, "codes": v.codes, "base_codes": base.codes, "stderr": v.stderr, "base_stderr": base.stderr }); + if v.status != base.status && v.status != Status::Timeout && base.status != Status::Timeout { + found.push(entry(format!("verdict: {:?} -> {:?}", base.status, v.status).to_lowercase())); + kept.push((format!("{name}.rs"), text.into_bytes())); + } else if v.status == Status::Error && base.status == Status::Error && v.codes != base.codes { + // Which error suppresses which may depend on order: a note. + notes.push(entry(format!("codes: {:?} -> {:?}", base.codes, v.codes))); + } + } + let label = |e: &serde_json::Value| format!("{}: {}", e["rewrite"].as_str().unwrap_or(""), e["what"].as_str().unwrap_or("")); + rec.found = found.iter().map(label).collect(); + rec.notes = notes.iter().map(label).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &kept, &serde_json::json!({ "kind": test.kind, "found": found, "notes": notes })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let names: Vec<&str> = match &args.rewrites { + Some(r) => r.split(',').collect(), + None => REWRITES.to_vec(), + }; + let tests = args.sweep.select(uitest::tests(&args.sweep.tests, uitest::ALL, |_| false)); + println!("{} tests, rewrites: {}", tests.len(), names.iter().filter(|n| **n != "identity").copied().collect::>().join(", ")); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &names, t))) +} diff --git a/crates/mirth-lab/src/tools/suggest_diff.rs b/crates/mirth-lab/src/tools/suggest_diff.rs new file mode 100644 index 0000000..ec38573 --- /dev/null +++ b/crates/mirth-lab/src/tools/suggest_diff.rs @@ -0,0 +1,193 @@ +//! Suggestions apply: a machine-applicable suggestion must produce code that compiles the way +//! the suggestion promises. +//! +//! For each standalone UI test without `//@ run-rustfix` (compiletest checks those), applies +//! each `MachineApplicable` suggestion inside the test file alone and compiles again: +//! +//! - lint-breaks: a warning's (a lint's) fix introduces an error; lints never stop a build, and +//! `cargo fix` applies their fixes without asking +//! - parse: the fixed file no longer parses +//! - not-fixed: not one fewer of the same diagnostic +//! - ice: the fixed file crashes the compiler +//! +//! Errors appearing after an error's suggestion is applied are expected and not reported. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{self, Compile, Diagnostic, Status}; +use mirth_lab::uitest::{self, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// Include tests with `//@ run-rustfix`. + #[arg(long)] + with_rustfix: bool, + /// Suggestions tried per test. + #[arg(long, default_value_t = 8)] + max: usize, + #[command(flatten)] + sweep: Sweep, +} + +static BY_PATH: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)^\s*(pub(\([^)]*\))?\s+)?mod\s+\w+\s*;|include(_str|_bytes)?!|#\[path").unwrap()); +static RUSTFIX: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^//@\s*run-rustfix").unwrap()); + +type Part = (usize, usize, String); + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + tried: usize, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn diagnose(args: &Args, test: &Test, source: &Path, out: &Path) -> (Status, Vec) { + let c = Compile::new(&args.rustc, source, out, &test.flags, test.edition()) + .emit("metadata") + .json() + .timeout(120) + .run(); + (c.status, rustc::diagnostics(&c.stderr)) +} + +/// The machine-applicable suggestions of a diagnostic (and its direct children) in `file`. +fn suggestions(diag: &Diagnostic, file: &str) -> Vec> { + std::iter::once(diag) + .chain(diag.children.iter()) + .filter_map(|node| { + let mut parts: Vec = node + .spans + .iter() + .filter(|s| { + s.suggestion_applicability.as_deref() == Some("MachineApplicable") + && Path::new(&s.file_name).file_name().and_then(|n| n.to_str()) == Some(file) + }) + .filter_map(|s| s.suggested_replacement.clone().map(|r| (s.byte_start, s.byte_end, r))) + .collect(); + parts.sort(); + (!parts.is_empty()).then_some(parts) + }) + .collect() +} + +fn key(d: &Diagnostic) -> (String, String) { + (d.code().to_owned(), d.message.clone()) +} + +/// The errors, by code (or lint name) when they have one: a renamed identifier changes the +/// message of the same lint. +fn errors(diags: &[Diagnostic]) -> BTreeSet<(String, String)> { + diags + .iter() + .filter(|d| d.level == "error" && !d.message.starts_with("aborting")) + .map(|d| if d.code().is_empty() { (String::new(), d.message.clone()) } else { (d.code().to_owned(), String::new()) }) + .collect() +} + +fn check(args: &Args, test: &Test) -> Rec { + let mut rec = Rec { test: test.rel.clone(), skip: None, tried: 0, found: Vec::new() }; + if BY_PATH.is_match(&test.text) { + rec.skip = Some("uses files by path".into()); + return rec; + } + let dir = driver::scratch_dir(&args.sweep); + let src = dir.path().join(test.file_name()); + let _ = std::fs::copy(&test.path, &src); + let (status, diags) = diagnose(args, test, &src, &dir.path().join("orig")); + if matches!(status, Status::Ice | Status::Timeout) { + rec.skip = Some(format!("original {status:?}").to_lowercase()); + return rec; + } + let base_errors = errors(&diags); + let text = std::fs::read(&src).unwrap_or_default(); + let mut found: Vec = Vec::new(); + let mut kept: Vec<(String, Vec)> = Vec::new(); + 'outer: for diag in &diags { + for parts in suggestions(diag, test.file_name()) { + if rec.tried >= args.max { + break 'outer; + } + rec.tried += 1; + // Apply from the end so earlier offsets stay valid; overlapping parts are skipped. + let mut fixed = text.clone(); + let mut last: Option = None; + let mut ok = true; + for (start, end, repl) in parts.iter().rev() { + if last.is_some_and(|l| *end > l) || *end > fixed.len() || start > end { + ok = false; + break; + } + fixed.splice(*start..*end, repl.bytes()); + last = Some(*start); + } + if !ok { + continue; + } + let fdir = dir.path().join(format!("fix{}", rec.tried)); + let _ = std::fs::create_dir_all(&fdir); + let fsrc = fdir.join(test.file_name()); + let _ = std::fs::write(&fsrc, &fixed); + let (fstatus, fdiags) = diagnose(args, test, &fsrc, &fdir); + let new: BTreeSet<_> = errors(&fdiags).difference(&base_errors).cloned().collect(); + let what = if fstatus == Status::Ice { + Some("ice") + } else if new.iter().any(|(_, m)| m.contains("expected") || m.contains("unexpected") || m.contains("unknown start of token")) { + Some("parse") + } else if diag.level == "warning" && !new.is_empty() { + Some("lint-breaks") + } else if fdiags.iter().filter(|d| key(d) == key(diag)).count() >= diags.iter().filter(|d| key(d) == key(diag)).count() { + // Nested braces legitimately report the next level, but then there is one fewer. + Some("not-fixed") + } else { + None + }; + if let Some(what) = what { + let name = format!("fix{}.rs", rec.tried); + found.push(serde_json::json!({ + "what": what, "diagnostic": diag.message, "code": diag.code(), "level": diag.level, + "parts": parts, "fixed_name": name, + "new_errors": new.iter().map(|(c, m)| if c.is_empty() { m.clone() } else { c.clone() }).take(5).collect::>(), + })); + kept.push((name, fixed)); + } + } + } + rec.found = found + .iter() + .map(|f| { + let who = if f["code"].as_str().unwrap_or("").is_empty() { f["level"].as_str().unwrap_or("") } else { f["code"].as_str().unwrap_or("") }; + format!("{}: {} {}", f["what"].as_str().unwrap_or(""), who, f["diagnostic"].as_str().unwrap_or("").chars().take(80).collect::()) + }) + .collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &kept, &serde_json::json!({ "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let tests = uitest::tests(&args.sweep.tests, uitest::ALL, |t| !args.with_rustfix && RUSTFIX.is_match(&t.text)); + let tests = args.sweep.select(tests); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, t))) +} From 91ff2721847fb72929345124bd85896f7455217b Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 07:29:44 +0000 Subject: [PATCH 09/23] mirth-lab: repro-diff, gate-check, instr-check (validated: same results as Python, instr-check with the harness artifacts it surfaced fixed); spawn retried on ETXTBSY Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- crates/mirth-lab/src/main.rs | 12 ++ crates/mirth-lab/src/rustc.rs | 39 +++- crates/mirth-lab/src/tools/gate_check.rs | 241 ++++++++++++++++++++++ crates/mirth-lab/src/tools/instr_check.rs | 201 ++++++++++++++++++ crates/mirth-lab/src/tools/repro_diff.rs | 137 ++++++++++++ 5 files changed, 625 insertions(+), 5 deletions(-) create mode 100644 crates/mirth-lab/src/tools/gate_check.rs create mode 100644 crates/mirth-lab/src/tools/instr_check.rs create mode 100644 crates/mirth-lab/src/tools/repro_diff.rs diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs index ee8423b..57fdfdf 100644 --- a/crates/mirth-lab/src/main.rs +++ b/crates/mirth-lab/src/main.rs @@ -8,8 +8,11 @@ use clap::{Parser, Subcommand}; mod tools { pub mod crash_diff; pub mod diag_check; + pub mod gate_check; + pub mod instr_check; pub mod miri_diff; pub mod opt_diff; + pub mod repro_diff; pub mod rewrite_diff; pub mod solver_diff; pub mod suggest_diff; @@ -38,6 +41,12 @@ enum Check { RewriteDiff(tools::rewrite_diff::Args), /// Machine-applicable suggestions, applied one at a time, keep the program compiling. SuggestDiff(tools::suggest_diff::Args), + /// Outputs depend only on inputs: repeat, other directory, threads, decoy libraries. + ReproDiff(tools::repro_diff::Args), + /// Nothing unstable is usable from stable code (attributes, library items). + GateCheck(tools::gate_check::Args), + /// PGO and coverage instrumentation round trips. + InstrCheck(tools::instr_check::Args), } fn main() -> ExitCode { @@ -50,6 +59,9 @@ fn main() -> ExitCode { Check::MiriDiff(a) => tools::miri_diff::run(a), Check::RewriteDiff(a) => tools::rewrite_diff::run(a), Check::SuggestDiff(a) => tools::suggest_diff::run(a), + Check::ReproDiff(a) => tools::repro_diff::run(a), + Check::GateCheck(a) => tools::gate_check::run(a), + Check::InstrCheck(a) => tools::instr_check::run(a), }; match result { Ok(code) => code, diff --git a/crates/mirth-lab/src/rustc.rs b/crates/mirth-lab/src/rustc.rs index 86c75ad..8952f3c 100644 --- a/crates/mirth-lab/src/rustc.rs +++ b/crates/mirth-lab/src/rustc.rs @@ -41,7 +41,18 @@ impl Finished { /// Run `cmd` with stdin closed, its output captured, killed after `timeout`. pub fn run_command(mut cmd: Command, timeout: Duration) -> std::io::Result { cmd.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); - let mut child = cmd.spawn()?; + // A binary just written can be "busy" when another thread forked while it was open for + // writing (the child holds the descriptor until it execs): retry for a while. + let mut tries = 0; + let mut child = loop { + match cmd.spawn() { + Err(e) if e.raw_os_error() == Some(26) && tries < 100 => { + tries += 1; + std::thread::sleep(Duration::from_millis(20)); + } + other => break other?, + } + }; // Read both pipes on threads, so a chatty child cannot block on a full pipe. let mut out = child.stdout.take().expect("piped"); let mut err = child.stderr.take().expect("piped"); @@ -114,6 +125,9 @@ pub struct Compile<'a> { pub timeout: Duration, /// Run with `RUSTC_BOOTSTRAP=1` (the default; a stable user's view needs it off). pub bootstrap: bool, + /// Name the output `/prog` with `-o` (the default); off when the extra options say + /// where outputs go (`--out-dir`). + pub name_output: bool, } impl<'a> Compile<'a> { @@ -129,6 +143,7 @@ impl<'a> Compile<'a> { json: false, timeout: Duration::from_secs(300), bootstrap: true, + name_output: true, } } @@ -147,6 +162,16 @@ impl<'a> Compile<'a> { self } + pub fn unnamed_output(mut self) -> Self { + self.name_output = false; + self + } + + pub fn stable(mut self) -> Self { + self.bootstrap = false; + self + } + pub fn timeout(mut self, secs: u64) -> Self { self.timeout = Duration::from_secs(secs); self @@ -158,10 +183,14 @@ impl<'a> Compile<'a> { let mut cmd = Command::new(self.rustc); cmd.arg(self.source) .args(["--edition", self.edition]) - .arg(format!("--emit={}", self.emit)) - .arg("-o") - .arg(&binary) - .args(["-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"]) + .arg(format!("--emit={}", self.emit)); + if self.name_output { + cmd.arg("-o").arg(&binary); + } + if self.bootstrap { + cmd.args(["-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"]); + } + cmd .arg(if self.json { "--error-format=json" } else { "--error-format=short" }) .args(self.flags) .args(&self.extra) diff --git a/crates/mirth-lab/src/tools/gate_check.rs b/crates/mirth-lab/src/tools/gate_check.rs new file mode 100644 index 0000000..38d3a83 --- /dev/null +++ b/crates/mirth-lab/src/tools/gate_check.rs @@ -0,0 +1,241 @@ +//! Feature gates: nothing unstable may be usable from stable code, whatever the spelling. +//! +//! Compiled without `#![feature]` and without `RUSTC_BOOTSTRAP`, as a stable user would: +//! +//! - attributes: every attribute in the "Unstable attributes" part of +//! compiler/rustc_feature/src/builtin_attrs.rs, on each kind of item and position +//! - library: every top-level `pub` item of core, alloc and std marked +//! `#[unstable(feature)]`, reached by `use` (direct, renamed, glob), implemented (traits), +//! taken as a value (functions) or named as a type +//! +//! A program must report the gate. One that compiles is a finding; one that fails without +//! mentioning the gate is noted. Library items whose path does not resolve even with the +//! feature are skipped. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::rustc::{Compile, Status}; +use rayon::prelude::*; +use regex::Regex; +use serde::Serialize; +use walkdir::WalkDir; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// The rust checkout (for the attribute table and the library sources). + #[arg(long)] + rust: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + jobs: usize, + /// `attributes` or `library`. + #[arg(long)] + only: Option, +} + +static GATE: LazyLock = LazyLock::new(|| { + Regex::new(r"E0658|is experimental|is unstable|unstable feature|use of unstable|internal implementation detail|unstable library feature|requires a nightly|used internally by the standard library|may not be used|are considered unstable|is an internal|cannot be used on stable").unwrap() +}); +/// Arguments for attributes that need them; the rest are written bare. +const ATTR_ARGS: &[(&str, &str)] = &[ + ("optimize", "(speed)"), ("patchable_function_entry", "(prefix_nops = 1, entry_nops = 1)"), + ("instrument_fn", " = \"on\""), ("cfi_encoding", " = \"u1x\""), ("register_tool", "(mytool)"), + ("register_attribute_tool", "(mytool)"), ("register_lint_tool", "(mytool)"), ("linkage", " = \"weak\""), + ("lang", " = \"mirth_nonexistent\""), ("rustc_on_unimplemented", "(message = \"x\")"), + ("rustc_diagnostic_item", " = \"mirth_x\""), ("test_runner", "(crate::r)"), ("pattern_complexity_limit", " = 10"), + ("rustc_legacy_const_generics", "(0)"), ("rustc_abi", "(debug)"), ("rustc_macro_transparency", " = \"semitransparent\""), + ("unstable", "(feature = \"x\", issue = \"none\")"), ("stable", "(feature = \"x\", since = \"1.0.0\")"), + ("rustc_const_unstable", "(feature = \"x\", issue = \"none\")"), ("rustc_const_stable", "(feature = \"x\", since = \"1.0.0\")"), + ("feature", "(mirth_nonexistent)"), ("rustc_objc_class", " = \"X\""), ("rustc_objc_selector", " = \"x\""), + ("rustc_confusables", "(\"x\")"), ("rustc_must_implement_one_of", "(a, b)"), ("allow_internal_unstable", "(core_intrinsics)"), + ("rustc_allow_const_fn_unstable", "(x)"), ("rustc_default_body_unstable", "(feature = \"x\", issue = \"none\")"), + ("rustc_simd_monomorphize_lane_limit", " = \"8\""), ("rustc_scalable_vector", "(4)"), +]; +/// Each position: a name, and a program with `{A}` where the outer attribute goes (`{AI}`: +/// the inner one). +const POSITIONS: &[(&str, &str)] = &[ + ("fn", "{A}\npub fn f() {}\nfn main() {}"), + ("fn-no-body", "pub trait T { {A} fn m(&self); }\nfn main() {}"), + ("foreign-fn", "unsafe extern \"C\" { {A} fn ext(); }\nfn main() {}"), + ("param", "pub fn f({A} x: u32) -> u32 { x }\nfn main() {}"), + ("param-no-body", "pub trait T { fn m(&self, {A} x: u32); }\nfn main() {}"), + ("fn-ptr-param", "pub type F = fn({A} u32);\nfn main() {}"), + ("struct", "{A}\npub struct S;\nfn main() {}"), + ("field", "pub struct S { {A} pub x: u32 }\nfn main() {}"), + ("impl", "pub struct S;\n{A}\nimpl S {}\nfn main() {}"), + ("trait", "{A}\npub trait T {}\nfn main() {}"), + ("mod", "{A}\npub mod m {}\nfn main() {}"), + ("closure", "fn main() { let _c = {A} || (); }"), + ("statement", "fn main() { {A} let _x = 1; }"), + ("crate", "#![{AI}]\nfn main() {}"), +]; + +#[derive(Serialize)] +struct Res { + kind: &'static str, + item: String, + position: String, + result: String, + first: String, +} + +fn compile_text(args: &Args, text: &str, feature: Option<&str>) -> (Status, String) { + let dir = tempfile::tempdir_in(&args.work).expect("scratch"); + let f = dir.path().join("t.rs"); + let body = match feature { + Some(feat) => format!("#![feature({feat})]\n{text}"), + None => text.to_owned(), + }; + let _ = std::fs::write(&f, body); + let mut c = Compile::new(&args.rustc, &f, dir.path(), &[], "2021").emit("metadata").timeout(60); + if feature.is_none() { + c = c.stable(); + } + let c = c.run(); + (c.status, c.stderr) +} + +fn classify(status: Status, stderr: &str, item: &str) -> String { + // `#[feature]` outside the crate root does nothing; rustc warns that it belongs at the root. + if status == Status::Ok && item == "feature" && stderr.contains("crate-level attribute") { + return "gated".into(); + } + match status { + Status::Ok => "accepted".into(), + Status::Ice => "ice".into(), + _ if GATE.is_match(stderr) => "gated".into(), + _ => "no-gate-message".into(), + } +} + +fn first_error(stderr: &str) -> String { + stderr.lines().find(|l| l.starts_with("error")).unwrap_or("").chars().take(200).collect() +} + +fn attributes(args: &Args) -> Vec { + let table = std::fs::read_to_string(args.rust.join("compiler/rustc_feature/src/builtin_attrs.rs")).unwrap_or_default(); + let start = table.find("Unstable attributes:").unwrap_or(0); + let re = Regex::new(r"sym::([a-z_0-9]+)").unwrap(); + let mut names: Vec = re.captures_iter(&table[start..]).map(|c| c[1].to_owned()).collect(); + names.sort(); + names.dedup(); + let args_of: BTreeMap<&str, &str> = ATTR_ARGS.iter().copied().collect(); + let jobs: Vec<(String, &str, String)> = names + .iter() + .flat_map(|name| { + let a = args_of.get(name.as_str()).copied().unwrap_or(""); + POSITIONS.iter().map(move |(pos, t)| { + (name.clone(), *pos, t.replace("{AI}", &format!("{name}{a}")).replace("{A}", &format!("#[{name}{a}]"))) + }) + }) + .collect(); + jobs.par_iter() + .map(|(name, pos, prog)| { + let (status, err) = compile_text(args, prog, None); + Res { kind: "attribute", item: name.clone(), position: pos.to_string(), result: classify(status, &err, name), first: first_error(&err) } + }) + .collect() +} + +struct Item { + path: String, + kind: String, + feature: String, + generic: bool, +} + +fn library_items(rust: &Path) -> Vec { + let unstable = Regex::new(r#"^#\[unstable\(feature = "([a-z_0-9]+)""#).unwrap(); + let decl = Regex::new(r#"^pub (?:const |unsafe |auto |extern "C" )*(struct|enum|trait|union|type|fn|const|static|macro) ([A-Za-z_][A-Za-z_0-9]*)(<)?"#).unwrap(); + let mut items = Vec::new(); + for krate in ["core", "alloc", "std"] { + let root = rust.join("library").join(krate).join("src"); + for entry in WalkDir::new(&root).into_iter().filter_map(Result::ok) { + let p = entry.path(); + if p.extension().is_none_or(|x| x != "rs") { + continue; + } + let rel = p.strip_prefix(&root).unwrap().with_extension(""); + let mut module = vec![krate.to_owned()]; + module.extend(rel.iter().map(|c| c.to_string_lossy().into_owned()).filter(|c| c != "lib" && c != "mod")); + let Ok(text) = std::fs::read_to_string(p) else { continue }; + let lines: Vec<&str> = text.lines().collect(); + for (i, line) in lines.iter().enumerate() { + let Some(m) = unstable.captures(line) else { continue }; + for next in lines.iter().skip(i + 1).take(5) { + if next.starts_with('#') { + continue; + } + if let Some(d) = decl.captures(next) { + items.push(Item { path: format!("{}::{}", module.join("::"), &d[2]), kind: d[1].to_owned(), feature: m[1].to_owned(), generic: d.get(3).is_some() }); + } + break; + } + } + } + } + items +} + +fn library(args: &Args) -> Vec { + library_items(&args.rust) + .par_iter() + .flat_map(|item| { + let (status, _) = compile_text(args, &format!("#[allow(unused_imports)] use {};\nfn main() {{}}", item.path), Some(&item.feature)); + if status != Status::Ok { + return vec![Res { kind: "library", item: item.path.clone(), position: "path".into(), result: "skipped: path".into(), first: String::new() }]; + } + let (parent, name) = item.path.rsplit_once("::").unwrap(); + let p = &item.path; + let mut progs = vec![ + ("use", format!("#[allow(unused_imports)] use {p};\nfn main() {{}}")), + ("use-as", format!("#[allow(unused_imports)] use {p} as Renamed;\nfn main() {{}}")), + ("glob", format!("#[allow(unused_imports)] use {parent}::*;\n#[allow(unused_imports)] use self::{name} as _;\nfn main() {{}}")), + ]; + if item.kind == "trait" && !item.generic { + progs.push(("impl", format!("struct L;\nimpl {p} for L {{}}\nfn main() {{}}"))); + } + if item.kind == "fn" && !item.generic { + progs.push(("value", format!("fn main() {{ let _f = {p}; }}"))); + } + if ["struct", "enum", "union", "type"].contains(&item.kind.as_str()) && !item.generic { + progs.push(("type", format!("pub fn g(_: Option<&{p}>) {{}}\nfn main() {{}}"))); + } + progs + .into_iter() + .map(|(pos, prog)| { + let (status, err) = compile_text(args, &prog, None); + Res { kind: "library", item: item.path.clone(), position: pos.into(), result: classify(status, &err, ""), first: first_error(&err) } + }) + .collect() + }) + .collect() +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build_global().ok(); + let mut results = Vec::new(); + if args.only.as_deref().is_none_or(|o| o == "attributes") { + results.extend(attributes(&args)); + } + if args.only.as_deref().is_none_or(|o| o == "library") { + results.extend(library(&args)); + } + std::fs::write(args.work.join("results.json"), serde_json::to_string_pretty(&results)?)?; + let mut counts: BTreeMap<(&str, &str), usize> = BTreeMap::new(); + for r in &results { + *counts.entry((r.kind, r.result.as_str())).or_default() += 1; + } + println!("{counts:?}"); + for r in results.iter().filter(|r| r.result == "accepted" || r.result == "ice") { + println!("{:9} {:9} {:45} {}", r.result.to_uppercase(), r.kind, r.item, r.position); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/instr_check.rs b/crates/mirth-lab/src/tools/instr_check.rs new file mode 100644 index 0000000..91158e9 --- /dev/null +++ b/crates/mirth-lab/src/tools/instr_check.rs @@ -0,0 +1,201 @@ +//! Instrumentation round trip: instrumented programs must behave as uninstrumented ones and +//! write profiles that LLVM's tools accept. +//! +//! For each runnable UI test, with a toolchain that ships the profiler runtime and llvm-tools: +//! PGO (`-Cprofile-generate`, run, `llvm-profdata merge`, `-Cprofile-use`, run) and coverage +//! (`-Cinstrument-coverage`, run, merge, `llvm-cov export`). Findings: behavior differing from +//! the plain build, no profile written, the LLVM tools failing or warning about corrupt data, +//! the compiler failing or crashing on `-Cprofile-use`. Threaded tests are left out (scheduling); +//! a test relying on the linker discarding an unused symbol (`linking/executable-no-mangle-strip`) +//! fails to link instrumented, as expected. + +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::Duration; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::normalize; +use mirth_lab::rustc::{self, Compile, Exit, Status, run_command}; +use mirth_lab::uitest::{self, Kind, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// A toolchain with the profiler runtime and llvm-tools. + #[arg(long)] + toolchain: String, + #[command(flatten)] + sweep: Sweep, +} + +static OWN: LazyLock = LazyLock::new(|| { + Regex::new(r"profile|instrument-coverage|coverage-options|^-O$|opt-level|panic=|prefer-dynamic|codegen-backend|-Clto|lto=|no-prepopulate").unwrap() +}); +static THREADS: LazyLock = LazyLock::new(|| Regex::new(r"thread::(spawn|scope)|std::sync::mpsc|\bspawn\(").unwrap()); +static BAD: LazyLock = LazyLock::new(|| Regex::new(r"(?i)corrupt|malformed|invalid|truncated|failed to|error").unwrap()); + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +struct Toolchain { + rustc: PathBuf, + tools: PathBuf, +} + +fn tool(tc: &Toolchain, name: &str, args: &[String]) -> (Exit, String) { + let mut cmd = Command::new(tc.tools.join(name)); + cmd.args(args); + match run_command(cmd, Duration::from_secs(120)) { + Ok(d) => { + let out = d.stdout_text(); + let text = d.stderr_text() + &out[out.len().saturating_sub(200)..]; + (d.exit, text) + } + Err(e) => (Exit::Code(-1), e.to_string()), + } +} + +fn observed(binary: &Path, env: &[(&str, &str)]) -> (Exit, String) { + // Every build runs from the same path (some tests print argv[0]); the test harness's + // timings and result order are normalized. + let fixed = binary.parent().and_then(Path::parent).unwrap_or(Path::new(".")).join("run").join("prog"); + let _ = std::fs::create_dir_all(fixed.parent().unwrap()); + let _ = std::fs::copy(binary, &fixed); + let o = rustc::observe(&fixed, 30, env); + (o.exit, normalize::stdout(&o.stdout)) +} + +fn profraws(dir: &Path) -> Vec { + std::fs::read_dir(dir) + .map(|d| d.flatten().filter(|e| e.path().extension().is_some_and(|x| x == "profraw")).map(|e| e.path().to_string_lossy().into_owned()).collect()) + .unwrap_or_default() +} + +fn tail(s: &str, n: usize) -> String { + let v: Vec = s.chars().collect(); + v[v.len().saturating_sub(n)..].iter().collect() +} + +fn check(args: &Args, tc: &Toolchain, test: &Test) -> Rec { + let mut rec = Rec { test: test.rel.clone(), skip: None, found: Vec::new() }; + let dir = driver::scratch_dir(&args.sweep); + let d = dir.path(); + let compile = |name: &str, extra: Vec| Compile::new(&tc.rustc, &test.path, &d.join(name), &test.flags, test.edition()).extra(extra).run(); + let plain = compile("plain", vec!["-Copt-level=1".into()]); + let Some(plain_bin) = plain.binary else { + rec.skip = Some(format!("plain build {:?}", plain.status).to_lowercase()); + return rec; + }; + let base = observed(&plain_bin, &[]); + if (0..2).any(|_| observed(&plain_bin, &[]) != base) { + rec.skip = Some("nondeterministic".into()); + return rec; + } + let mut found: Vec = Vec::new(); + // PGO: generate, merge, use. + let raw = d.join("pgo-raw"); + let generated = compile("gen", vec!["-Copt-level=1".into(), format!("-Cprofile-generate={}", raw.display())]); + match generated.binary { + None => found.push(serde_json::json!({ "what": format!("profile-generate build {:?}", generated.status).to_lowercase(), "stderr": tail(&generated.stderr, 1500) })), + Some(bin) => { + let got = observed(&bin, &[]); + if got != base { + found.push(serde_json::json!({ "what": "profile-generate changes behavior", "base": base, "got": got })); + } + let raws = profraws(&raw); + if raws.is_empty() && got.0 == Exit::Code(0) { + found.push(serde_json::json!({ "what": "no raw profile written" })); + } else if !raws.is_empty() { + let merged = d.join("merged.profdata"); + let mut a = vec!["merge".to_string(), "-o".into(), merged.to_string_lossy().into_owned()]; + a.extend(raws); + let (exit, out) = tool(tc, "llvm-profdata", &a); + if exit != Exit::Code(0) || BAD.is_match(&out) { + found.push(serde_json::json!({ "what": "llvm-profdata merge", "exit": exit, "out": tail(&out, 1500) })); + } else { + let used = compile("use", vec!["-Copt-level=2".into(), format!("-Cprofile-use={}", merged.display())]); + match (used.status, used.binary) { + (Status::Ice, _) => found.push(serde_json::json!({ "what": "profile-use ICE", "stderr": tail(&used.stderr, 2000) })), + (s, None) => found.push(serde_json::json!({ "what": format!("profile-use build {s:?}").to_lowercase(), "stderr": tail(&used.stderr, 1500) })), + (_, Some(ub)) => { + let got = observed(&ub, &[]); + if got != base { + found.push(serde_json::json!({ "what": "profile-use changes behavior", "base": base, "got": got })); + } + } + } + } + } + } + } + // Coverage: instrument, merge, export. + let cov = compile("cov", vec!["-Cinstrument-coverage".into()]); + match cov.binary { + None => found.push(serde_json::json!({ "what": format!("instrument-coverage build {:?}", cov.status).to_lowercase(), "stderr": tail(&cov.stderr, 1500) })), + Some(bin) => { + let craw = d.join("cov-raw"); + let _ = std::fs::create_dir_all(&craw); + let pattern = craw.join("c-%p.profraw").to_string_lossy().into_owned(); + let got = observed(&bin, &[("LLVM_PROFILE_FILE", &pattern)]); + if got != base { + found.push(serde_json::json!({ "what": "instrument-coverage changes behavior", "base": base, "got": got })); + } + let raws = profraws(&craw); + if !raws.is_empty() { + let merged = d.join("cov.profdata"); + let mut a = vec!["merge".to_string(), "-sparse".into(), "-o".into(), merged.to_string_lossy().into_owned()]; + a.extend(raws); + let (exit, out) = tool(tc, "llvm-profdata", &a); + if exit != Exit::Code(0) || BAD.is_match(&out) { + found.push(serde_json::json!({ "what": "llvm-profdata merge (coverage)", "exit": exit, "out": tail(&out, 1500) })); + } else { + let a = vec!["export".to_string(), "-summary-only".into(), format!("-instr-profile={}", merged.display()), bin.to_string_lossy().into_owned()]; + let (exit, out) = tool(tc, "llvm-cov", &a); + // A binary with nothing to instrument (a test harness without tests) has an + // empty coverage map, which llvm-cov refuses. + if exit != Exit::Code(0) && !out.contains("no coverage data found") { + found.push(serde_json::json!({ "what": "llvm-cov export", "exit": exit, "out": tail(&out, 1500) })); + } + } + } else if got.0 == Exit::Code(0) { + found.push(serde_json::json!({ "what": "no coverage profile written" })); + } + } + } + rec.found = found.iter().map(|f| f["what"].as_str().unwrap_or("").to_owned()).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let home = PathBuf::from(std::env::var("HOME")?); + let root = home.join(format!(".rustup/toolchains/{}-x86_64-unknown-linux-gnu", args.toolchain)); + let tc = Toolchain { rustc: root.join("bin/rustc"), tools: root.join("lib/rustlib/x86_64-unknown-linux-gnu/bin") }; + // Threaded tests' output order depends on scheduling, which instrumentation changes. + // Output that depends on a random hash seed. + const NOISE: &[&str] = &["collections/hashmap/hashmap-debug-format.rs"]; + let tests = uitest::tests(&args.sweep.tests, &[Some(Kind::RunPass)], |t| { + uitest::flag_matches(t, &OWN) || THREADS.is_match(&t.text) || NOISE.contains(&t.rel.as_str()) + }); + let tests = args.sweep.select(tests); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &tc, t))) +} diff --git a/crates/mirth-lab/src/tools/repro_diff.rs b/crates/mirth-lab/src/tools/repro_diff.rs new file mode 100644 index 0000000..0d7ca21 --- /dev/null +++ b/crates/mirth-lab/src/tools/repro_diff.rs @@ -0,0 +1,137 @@ +//! Determinism: what rustc writes must depend only on its inputs and options. +//! +//! Builds each standalone UI test that compiles several times and compares the outputs +//! (`.rmeta`, `.rlib` normalized member by member, executables) with the first build: +//! +//! - repeat: the same build again, in the same directory +//! - path: the same build in another directory, both with `--remap-path-prefix` to one name +//! - threads: `-Zthreads=8` (tests marked `ignore-parallel-frontend` skip it); known: async fns +//! (rust-lang/rust#162202), RPIT and impl Trait in traits (#163878) +//! - decoy: a `-L` directory holding unrelated libraries whose names start with the crate's +//! name (#159677's shape) + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::artifacts; +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{Compile, Status}; +use mirth_lab::uitest::{self, Kind, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// Comma-separated subset of repeat, path, threads, decoy. + #[arg(long)] + variants: Option, + #[command(flatten)] + sweep: Sweep, +} + +const VARIANTS: &[&str] = &["repeat", "path", "threads", "decoy"]; +static OWN: LazyLock = + LazyLock::new(|| Regex::new(r"threads|remap-path|-o\b|--out-dir|emit|crate-name|extern|-L\b|-Cincremental").unwrap()); +static PARALLEL_IGNORED: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^//@\s*ignore-parallel-frontend").unwrap()); + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +/// Build the test copied into `src_dir`; its outputs as {file: digest}, or None if it fails. +fn build(args: &Args, test: &Test, src_dir: &Path, extra: &[String]) -> Option> { + let out = src_dir.join("out"); + let _ = std::fs::remove_dir_all(&out); + let _ = std::fs::create_dir_all(&out); + let emit = if test.kind == Some(Kind::CheckPass) { "metadata" } else { "link,metadata" }; + let source = src_dir.join(test.file_name()); + let mut extra_all = vec![ + "--out-dir".to_string(), + out.to_string_lossy().into_owned(), + "--crate-name".into(), + "t".into(), + format!("--remap-path-prefix={}=/src", src_dir.display()), + ]; + extra_all.extend_from_slice(extra); + let c = Compile::new(&args.rustc, &source, &out, &test.flags, test.edition()).extra(extra_all).emit(emit).unnamed_output().run(); + (c.status == Status::Ok).then(|| artifacts::digest_dir(&out)) +} + +fn check(args: &Args, variants: &[&str], test: &Test) -> Rec { + let mut rec = Rec { test: test.rel.clone(), skip: None, found: Vec::new() }; + let dir = driver::scratch_dir(&args.sweep); + let (a, b) = (dir.path().join("a"), dir.path().join("elsewhere-b")); + for d in [&a, &b] { + let _ = std::fs::create_dir_all(d); + let _ = std::fs::copy(&test.path, d.join(test.file_name())); + } + let Some(base) = build(args, test, &a, &[]) else { + rec.skip = Some("does not build".into()); + return rec; + }; + let mut found: Vec<(String, String)> = Vec::new(); + for &v in variants { + if v == "threads" && PARALLEL_IGNORED.is_match(&test.text) { + continue; + } + let got = match v { + "repeat" => build(args, test, &a, &[]), + "path" => build(args, test, &b, &[]), + "threads" => build(args, test, &a, &["-Zthreads=8".into()]), + "decoy" => { + let decoy = dir.path().join("decoy"); + let _ = std::fs::create_dir_all(&decoy); + let _ = std::fs::write(decoy.join("libtother.rlib"), b"!\n"); + let _ = std::fs::write(decoy.join("libt-0123456789abcdef.rmeta"), b"rust\0\0\0\0"); + build(args, test, &a, &["-L".into(), decoy.to_string_lossy().into_owned()]) + } + _ => continue, + }; + match got { + None => found.push((v.into(), "does not build".into())), + Some(g) if g != base => { + let differ: Vec<&str> = g.keys().chain(base.keys()).filter(|k| g.get(*k) != base.get(*k)).map(String::as_str).collect::>().into_iter().collect(); + found.push((v.into(), format!("outputs differ: {}", differ.join(", ")))); + } + _ => {} + } + } + // A difference also in `repeat` is the build's own nondeterminism: report only that. + if found.iter().any(|(v, _)| v == "repeat") { + found.retain(|(v, _)| v == "repeat"); + } + rec.found = found.iter().map(|(v, w)| format!("{v}: {w}")).collect(); + if !found.is_empty() { + let detail: Vec<_> = found.iter().map(|(v, w)| serde_json::json!({ "variant": v, "what": w })).collect(); + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "found": detail })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let variants: Vec<&str> = match &args.variants { + Some(v) => v.split(',').collect(), + None => VARIANTS.to_vec(), + }; + let kinds = [Some(Kind::BuildPass), Some(Kind::RunPass), Some(Kind::CheckPass)]; + let tests = args.sweep.select(uitest::tests(&args.sweep.tests, &kinds, |t| uitest::flag_matches(t, &OWN))); + println!("{} tests, variants: {}", tests.len(), variants.join(", ")); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &variants, t))) +} From 82f27a9ac197ddf0eb6c42f1c5c2ae626383a741 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 07:45:31 +0000 Subject: [PATCH 10/23] mirth-lab: release-diff, xlink, scale-check (wait4 rusage through libc), abi-diff (typed differences; known and undecided classes labelled by variant); all 14 oracle tools ported Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- crates/mirth-lab/Cargo.toml | 1 + crates/mirth-lab/src/driver.rs | 7 +- crates/mirth-lab/src/main.rs | 16 + crates/mirth-lab/src/normalize.rs | 6 +- crates/mirth-lab/src/tools/abi_diff.rs | 644 +++++++++++++++++++++ crates/mirth-lab/src/tools/release_diff.rs | 158 +++++ crates/mirth-lab/src/tools/scale_check.rs | 282 +++++++++ crates/mirth-lab/src/tools/xlink.rs | 173 ++++++ 8 files changed, 1285 insertions(+), 2 deletions(-) create mode 100644 crates/mirth-lab/src/tools/abi_diff.rs create mode 100644 crates/mirth-lab/src/tools/release_diff.rs create mode 100644 crates/mirth-lab/src/tools/scale_check.rs create mode 100644 crates/mirth-lab/src/tools/xlink.rs diff --git a/crates/mirth-lab/Cargo.toml b/crates/mirth-lab/Cargo.toml index 149e5ca..83fe69a 100644 --- a/crates/mirth-lab/Cargo.toml +++ b/crates/mirth-lab/Cargo.toml @@ -21,4 +21,5 @@ sha2 = "0.10" tempfile = "3" walkdir = "2" wait-timeout = "0.2" +libc = "0.2" mirth-rewrite = { path = "../mirth-rewrite" } diff --git a/crates/mirth-lab/src/driver.rs b/crates/mirth-lab/src/driver.rs index d8a7c2e..673f6bf 100644 --- a/crates/mirth-lab/src/driver.rs +++ b/crates/mirth-lab/src/driver.rs @@ -121,7 +121,12 @@ pub fn drive(items: &[T], sweep: &Sweep, check: impl Fn(&T) let done = AtomicUsize::new(0); let with_findings = AtomicUsize::new(0); let total = items.len(); - let pool = rayon::ThreadPoolBuilder::new().num_threads(sweep.jobs.max(1)).build().expect("thread pool"); + // Large stacks: in-process parsers (syn in the rewrites) recurse as deep as a test nests. + let pool = rayon::ThreadPoolBuilder::new() + .num_threads(sweep.jobs.max(1)) + .stack_size(256 << 20) + .build() + .expect("thread pool"); pool.install(|| { items.par_iter().for_each(|item| { if stop.load(Ordering::Relaxed) { diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs index 57fdfdf..6a82e55 100644 --- a/crates/mirth-lab/src/main.rs +++ b/crates/mirth-lab/src/main.rs @@ -6,16 +6,20 @@ use std::process::ExitCode; use clap::{Parser, Subcommand}; mod tools { + pub mod abi_diff; pub mod crash_diff; pub mod diag_check; pub mod gate_check; pub mod instr_check; pub mod miri_diff; pub mod opt_diff; + pub mod release_diff; pub mod repro_diff; pub mod rewrite_diff; + pub mod scale_check; pub mod solver_diff; pub mod suggest_diff; + pub mod xlink; } #[derive(Parser)] @@ -47,6 +51,14 @@ enum Check { GateCheck(tools::gate_check::Args), /// PGO and coverage instrumentation round trips. InstrCheck(tools::instr_check::Args), + /// Real crates accepted by one toolchain are accepted by the next, in comparable time. + ReleaseDiff(tools::release_diff::Args), + /// Every target builds core and alloc and links a program with no undefined symbols. + Xlink(tools::xlink::Args), + /// Compile time, memory, frames and future sizes grow about linearly with program size. + ScaleCheck(tools::scale_check::Args), + /// rustc's extern "C" lowering matches clang's for random C signatures, per target. + AbiDiff(tools::abi_diff::Args), } fn main() -> ExitCode { @@ -62,6 +74,10 @@ fn main() -> ExitCode { Check::ReproDiff(a) => tools::repro_diff::run(a), Check::GateCheck(a) => tools::gate_check::run(a), Check::InstrCheck(a) => tools::instr_check::run(a), + Check::ReleaseDiff(a) => tools::release_diff::run(a), + Check::Xlink(a) => tools::xlink::run(a), + Check::ScaleCheck(a) => tools::scale_check::run(a), + Check::AbiDiff(a) => tools::abi_diff::run(a), }; match result { Ok(code) => code, diff --git a/crates/mirth-lab/src/normalize.rs b/crates/mirth-lab/src/normalize.rs index 7180168..bb85aa9 100644 --- a/crates/mirth-lab/src/normalize.rs +++ b/crates/mirth-lab/src/normalize.rs @@ -16,7 +16,11 @@ static TIMING: LazyLock = LazyLock::new(|| Regex::new(r"finished in \d+\. pub fn stderr(text: &str) -> String { let lines: Vec<&str> = text .lines() - .filter(|l| !l.starts_with("note: run with `RUST_BACKTRACE") && !l.starts_with("note: Some details are omitted")) + .filter(|l| { + !l.starts_with("note: run with `RUST_BACKTRACE") + && !l.starts_with("note: Some details are omitted") + && !l.starts_with("note: in Miri, you may have to set `MIRIFLAGS") + }) .collect(); let t = THREAD_ID.replace_all(&lines.join("\n"), "$1").into_owned(); let t = STD_PATH.replace_all(&t, "library/").into_owned(); diff --git a/crates/mirth-lab/src/tools/abi_diff.rs b/crates/mirth-lab/src/tools/abi_diff.rs new file mode 100644 index 0000000..1100ce8 --- /dev/null +++ b/crates/mirth-lab/src/tools/abi_diff.rs @@ -0,0 +1,644 @@ +//! ABI differential: rustc's `extern "C"` must lower a signature the way clang lowers the same +//! C signature, on every target both support. +//! +//! Generates random C signatures (bool, integers of each width, float, double, pointers, +//! `__int128` on 64-bit targets, and repr(C) structs, unions and arrays of them, nested, packed +//! or over-aligned), writes each as a C function (clang, the target's LLVM triple, CPU and +//! features) and a Rust `#[no_mangle] extern "C" fn` (rustc against minicore, no sysroot), and +//! compares the two LLVM IR signatures parameter by parameter after first-class aggregates are +//! flattened. A difference in register class, extension, inreg/byval/sret, byval alignment, +//! parameter count or calling convention is a finding; representation-only differences are +//! notes. Verified equivalences per architecture, known bugs (#163911, findings 19 and 20) and +//! two differences this host cannot decide are labelled. + +use std::collections::BTreeMap; +use std::path::PathBuf; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; + +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// The rust checkout (for tests/auxiliary/minicore.rs). + #[arg(long)] + rust: PathBuf, + #[arg(long)] + work: PathBuf, + /// Comma-separated targets (default: the main tier 1 and 2 targets). + #[arg(long)] + targets: Option, + /// Every target rustc knows (the non-main ones still show representation differences). + #[arg(long)] + all: bool, + #[arg(long, default_value_t = 200)] + count: usize, + #[arg(long, default_value_t = 1)] + seed: u64, + #[arg(long, default_value_t = 8)] + jobs: usize, + #[arg(long, default_value = "clang")] + clang: String, +} + +const MAIN: &[&str] = &[ + "x86_64-unknown-linux-gnu", "x86_64-unknown-linux-musl", "x86_64-pc-windows-msvc", "x86_64-pc-windows-gnu", + "x86_64-apple-darwin", "i686-unknown-linux-gnu", "i686-pc-windows-msvc", "aarch64-unknown-linux-gnu", + "aarch64-apple-darwin", "aarch64-pc-windows-msvc", "aarch64-unknown-linux-musl", "armv7-unknown-linux-gnueabihf", + "arm-unknown-linux-gnueabi", "thumbv7em-none-eabihf", "riscv64gc-unknown-linux-gnu", "riscv32imac-unknown-none-elf", + "loongarch64-unknown-linux-gnu", "powerpc64le-unknown-linux-gnu", "s390x-unknown-linux-gnu", "wasm32-unknown-unknown", + "wasm32-wasip1", +]; + +// ---- generation ---- + +/// splitmix64: a small deterministic generator, so a seed names a program. +struct Rng(u64); +impl Rng { + fn next(&mut self) -> u64 { + self.0 = self.0.wrapping_add(0x9e37_79b9_7f4a_7c15); + let mut z = self.0; + z = (z ^ (z >> 30)).wrapping_mul(0xbf58_476d_1ce4_e5b9); + z = (z ^ (z >> 27)).wrapping_mul(0x94d0_49bb_1331_11eb); + z ^ (z >> 31) + } + fn float(&mut self) -> f64 { + (self.next() >> 11) as f64 / (1u64 << 53) as f64 + } + fn below(&mut self, n: usize) -> usize { + (self.next() % n as u64) as usize + } + fn pick<'a, T>(&mut self, v: &'a [T]) -> &'a T { + &v[self.below(v.len())] + } +} + +const SCALARS: &[(&str, &str)] = &[ + ("_Bool", "bool"), ("signed char", "i8"), ("unsigned char", "u8"), ("short", "i16"), ("unsigned short", "u16"), + ("int", "i32"), ("unsigned int", "u32"), ("long long", "i64"), ("unsigned long long", "u64"), ("float", "f32"), + ("double", "f64"), ("void*", "*mut u8"), +]; +const WIDE: &[(&str, &str)] = &[("__int128", "i128"), ("unsigned __int128", "u128")]; + +#[derive(Clone)] +enum Field { + Scalar(String, String), + Array(String, String, usize), + Named(String), +} + +struct Aggregate { + name: String, + union: bool, + packed: bool, + align: Option, + fields: Vec, +} + +struct Gen { + rng: Rng, + wide: bool, + aggregates: Vec, + names: usize, +} + +impl Gen { + fn scalar(&mut self) -> (String, String) { + let pool: Vec<(&str, &str)> = SCALARS.iter().chain(if self.wide { WIDE } else { &[] }).copied().collect(); + let (c, r) = *self.rng.pick(&pool); + (c.into(), r.into()) + } + fn field(&mut self, depth: u32) -> Field { + let r = self.rng.float(); + if depth < 2 && r < 0.15 { + return Field::Named(self.aggregate(depth + 1)); + } + let (c, rs) = self.scalar(); + if r < 0.25 { + let n = *self.rng.pick(&[1, 2, 3, 4, 8]); + return Field::Array(c, rs, n); + } + Field::Scalar(c, rs) + } + fn aggregate(&mut self, depth: u32) -> String { + let name = format!("S{}", self.names); + self.names += 1; + let union = self.rng.float() < 0.12; + let packed = !union && self.rng.float() < 0.08; + // Rust rejects packed with align, and a packed type holding an over-aligned one. + let align = if packed { None } else { *self.rng.pick(&[None, None, None, None, None, None, None, None, None, Some(16), Some(32)]) }; + let n = 1 + self.rng.below(5); + let fields = (0..n).map(|_| self.field(if packed { 2 } else { depth })).collect(); + self.aggregates.push(Aggregate { name: name.clone(), union, packed, align, fields }); + name + } + /// (C type, Rust type) + fn ty(&mut self) -> (String, String) { + if self.rng.float() < 0.4 { + let n = self.aggregate(0); + (n.clone(), n) + } else { + self.scalar() + } + } +} + +fn program(wide: bool, seed: u64, count: usize) -> (String, String, Vec) { + let mut g = Gen { rng: Rng(seed), wide, aggregates: Vec::new(), names: 0 }; + let mut fns = Vec::new(); + for k in 0..count { + let n = g.rng.below(9); + let params: Vec<(String, String)> = (0..n).map(|_| g.ty()).collect(); + let ret = if g.rng.float() < 0.15 { None } else { Some(g.ty()) }; + fns.push((format!("f{k}"), params, ret)); + } + let mut c = vec!["#include ".to_owned()]; + let mut rs: Vec = [ + "#![feature(no_core)]", "#![no_core]", "#![crate_type = \"lib\"]", + "#![allow(improper_ctypes_definitions, unused, non_snake_case)]", "extern crate minicore;", "use minicore::*;", + ] + .iter() + .map(|s| s.to_string()) + .collect(); + for a in &g.aggregates { + let kw = if a.union { "union" } else { "struct" }; + let cf: Vec = a.fields.iter().enumerate().map(|(i, f)| match f { + Field::Scalar(ct, _) => format!("{ct} f{i};"), + Field::Array(ct, _, n) => format!("{ct} f{i}[{n}];"), + Field::Named(n) => format!("{n} f{i};"), + }).collect(); + let rf: Vec = a.fields.iter().enumerate().map(|(i, f)| match f { + Field::Scalar(_, rt) => format!("pub f{i}: {rt},"), + Field::Array(_, rt, n) => format!("pub f{i}: [{rt}; {n}],"), + Field::Named(n) => format!("pub f{i}: {n},"), + }).collect(); + let cattr = format!("{}{}", if a.packed { " __attribute__((packed))" } else { "" }, a.align.map(|x| format!(" __attribute__((aligned({x})))")).unwrap_or_default()); + c.push(format!("typedef {kw} {} {{ {} }}{cattr} {};", a.name, cf.join(" "), a.name)); + let repr = format!("C{}{}", if a.packed { ", packed" } else { "" }, a.align.map(|x| format!(", align({x})")).unwrap_or_default()); + rs.push(format!("#[repr({repr})] pub {kw} {} {{ {} }}", a.name, rf.join(" "))); + // Union fields must be Copy; minicore has no derive. + rs.push(format!("impl Copy for {} {{}}", a.name)); + } + let mut names = Vec::new(); + for (name, params, ret) in &fns { + let cp = if params.is_empty() { "void".into() } else { params.iter().enumerate().map(|(i, t)| format!("{} a{i}", t.0)).collect::>().join(", ") }; + c.push(format!("{} {name}({cp}) {{ for (;;); }}", ret.as_ref().map_or("void", |r| r.0.as_str()))); + let rp = params.iter().enumerate().map(|(i, t)| format!("a{i}: {}", t.1)).collect::>().join(", "); + let rr = ret.as_ref().map(|r| format!(" -> {}", r.1)).unwrap_or_default(); + rs.push(format!("#[no_mangle] pub extern \"C\" fn {name}({rp}){rr} {{ loop {{}} }}")); + names.push(name.clone()); + } + (c.join("\n") + "\n", rs.join("\n") + "\n", names) +} + +// ---- reading LLVM IR signatures ---- + +#[derive(Clone, Debug, PartialEq)] +struct Param { + ty: String, + attrs: Vec, + noundef: bool, +} + +#[derive(Clone, Debug)] +struct Sig { + cc: Vec, + ret: Vec, + ret_attrs: Vec, + params: Vec, +} + +static DEFINE: LazyLock = LazyLock::new(|| Regex::new(r"^define\s+(.*?)@(\w+)\((.*)\)(.*)\{\s*$").unwrap()); +static DROP: LazyLock = LazyLock::new(|| { + Regex::new(r"\b(noundef|nonnull|noalias|nocapture|readonly|readnone|writeonly|writable|dead_on_unwind|captures\([^)]*\)|dereferenceable(_or_null)?\(\d+\)|initializes\([^)]*\)|range\([^)]*\)|nofpclass\([^)]*\)|immarg|returned|local_unnamed_addr|unnamed_addr|dso_local|dso_preemptable|hidden|protected|internal|private|nounwind|noinline|optnone|!\w+ !\d+|#\d+)\b").unwrap() +}); +static NAMED: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^(%[\w.]+) = type (.*)$").unwrap()); +static CC: LazyLock = LazyLock::new(|| Regex::new(r"\b(\w+cc|cc \d+)\b").unwrap()); +static RET_TY: LazyLock = LazyLock::new(|| Regex::new(r"(\{[^{}]*\}|<\{[^{}]*\}>|\[[^\]]*\]|<[^>]*>|[\w.%*]+)\s*$").unwrap()); +static PARAM_NAME: LazyLock = LazyLock::new(|| Regex::new(r"\s+%[\w.]+$").unwrap()); +static PARAM_TY: LazyLock = LazyLock::new(|| Regex::new(r"^(\{[^}]*\}|\[[^\]]*\]|<[^>]*>|[\w.%*]+)(.*)$").unwrap()); +static ABI_ATTR: LazyLock = LazyLock::new(|| { + Regex::new(r"(zeroext|signext|inreg|byval\([^)]*\)|sret\([^)]*\)|byref\([^)]*\)|align \d+|inalloca\([^)]*\))").unwrap() +}); +static ONE_ELEM: LazyLock = LazyLock::new(|| Regex::new(r"^\[1 x (.*)\]$").unwrap()); +static ARRAY: LazyLock = LazyLock::new(|| Regex::new(r"^\[(\d+) x (i\d+|float|double)\]$").unwrap()); +static INT: LazyLock = LazyLock::new(|| Regex::new(r"^i(\d+)$").unwrap()); +static INT_ARRAY: LazyLock = LazyLock::new(|| Regex::new(r"^\[\d+ x i\d+\]$").unwrap()); + +fn split_top(s: &str) -> Vec { + let (mut out, mut depth, mut cur) = (Vec::new(), 0i32, String::new()); + for ch in s.chars() { + match ch { + '(' | '{' | '[' | '<' => depth += 1, + ')' | '}' | ']' | '>' => depth -= 1, + _ => {} + } + if ch == ',' && depth == 0 { + out.push(cur.trim().to_owned()); + cur.clear(); + } else { + cur.push(ch); + } + } + if !cur.trim().is_empty() { + out.push(cur.trim().to_owned()); + } + out +} + +fn param(p: &str) -> (String, Vec, bool) { + let p = PARAM_NAME.replace(p.trim(), "").into_owned(); + let noundef = p.contains("noundef"); + let (ty, attrs) = match PARAM_TY.captures(&p) { + Some(c) => (c[1].to_owned(), c[2].to_owned()), + None => (p.clone(), String::new()), + }; + let attrs = DROP.replace_all(&attrs, ""); + let mut kept: Vec = ABI_ATTR + .find_iter(&attrs) + .map(|m| { + let a = m.as_str(); + if ["byval", "sret", "byref", "inalloca"].iter().any(|k| a.starts_with(k)) { a.split('(').next().unwrap().to_owned() } else { a.to_owned() } + }) + .collect(); + // An `align` on a plain pointer is a hint, not ABI; on byval and sret it is ABI. + if !kept.iter().any(|k| k == "byval" || k == "sret" || k == "byref") { + kept.retain(|k| !k.starts_with("align")); + } + kept.sort(); + (ty, kept, noundef) +} + +fn flatten(ty: &str, named: &BTreeMap) -> Vec { + let mut ty = ty.trim().to_owned(); + if let Some(body) = named.get(&ty) { + ty = body.clone(); + } + if ty.starts_with("<{") && ty.ends_with("}>") { + ty = ty[1..ty.len() - 1].to_owned(); + } + if ty.starts_with('{') && ty.ends_with('}') { + return split_top(&ty[1..ty.len() - 1]).iter().flat_map(|p| flatten(p, named)).collect(); + } + if let Some(c) = ONE_ELEM.captures(&ty) { + return flatten(&c[1], named); + } + vec![ty] +} + +fn signatures(ll: &str) -> BTreeMap { + let named: BTreeMap = NAMED.captures_iter(ll).map(|c| (c[1].to_owned(), c[2].trim().to_owned())).collect(); + let mut out = BTreeMap::new(); + for line in ll.lines().filter(|l| l.starts_with("define")) { + let Some(c) = DEFINE.captures(line) else { continue }; + let cc: Vec = CC.find_iter(&c[1]).map(|m| m.as_str().to_owned()).collect(); + let head = DROP.replace_all(&c[1], "").trim().to_owned(); + let (ret_ty, prefix) = match RET_TY.find(&head) { + Some(m) => (m.as_str().trim().to_owned(), head[..m.start()].to_owned()), + None => ("void".to_owned(), String::new()), + }; + let mut ret_attrs: Vec = prefix.split_whitespace().filter(|a| ["zeroext", "signext", "inreg"].contains(a)).map(str::to_owned).collect(); + ret_attrs.sort(); + let params = split_top(&c[3]) + .iter() + .flat_map(|p| { + let (ty, attrs, noundef) = param(p); + flatten(&ty, &named).into_iter().map(move |t| Param { ty: t, attrs: attrs.clone(), noundef }) + }) + .collect(); + out.insert(c[2].to_owned(), Sig { cc, ret: flatten(&ret_ty, &named), ret_attrs, params }); + } + out +} + +// ---- comparing ---- + +fn klass(ty: &str) -> String { + match ty { + "float" | "double" | "half" | "bfloat" | "fp128" | "x86_fp80" | "ppc_fp128" => "fp".into(), + "void" => "void".into(), + _ if ty.starts_with('<') => "vec".into(), + _ if ty.starts_with('[') => { + let inner = ty.trim_start_matches('[').trim_end_matches(']'); + match inner.split_once(" x ") { + Some((n, e)) => format!("[{n} x {}]", klass(e)), + None => ty.into(), + } + } + _ => "int".into(), + } +} + +/// Register classes, one per register-sized unit. +fn units(types: &[String], arch: &str, ret: bool, width: usize) -> Vec { + let mut out = Vec::new(); + for t in types { + if let Some(c) = INT.captures(t) + && c[1].parse::().unwrap_or(0) > width + { + out.extend(std::iter::repeat_n("int".to_owned(), c[1].parse::().unwrap() / width)); + } else if let Some(c) = ARRAY.captures(t) + && (ret || (arch == "arm" && (&c[2] == "float" || &c[2] == "double"))) + { + // A float array is a homogeneous aggregate: in a return, or an ARM VFP argument, it + // takes consecutive registers like a struct of its elements. + let elems: Vec = std::iter::repeat_n(c[2].to_owned(), c[1].parse().unwrap_or(0)).collect(); + out.extend(units(&elems, arch, ret, width)); + } else if t == "agg" { + out.push("agg".into()); + } else { + let mut k = klass(t); + // x86's SSE registers hold floats, doubles and vectors alike. + if (arch == "x86_64" || arch == "x86") && (k == "fp" || k == "vec") { + k = "sse".into(); + } + out.push(k); + } + } + out +} + +#[derive(Debug, Clone, Serialize)] +#[serde(tag = "kind", rename_all = "kebab-case")] +enum Diff { + CallingConvention { clang: Vec, rustc: Vec }, + ReturnAttributes { clang: Vec, rustc: Vec, ret: Vec }, + Return { clang: Vec, rustc: Vec }, + Parameters { clang: Vec, rustc: Vec }, + ParameterAttributes { index: usize, clang: Vec, rustc: Vec, ty: String }, +} + +impl Diff { + fn label(&self) -> &'static str { + match self { + Diff::CallingConvention { .. } => "calling convention", + Diff::ReturnAttributes { .. } => "return attributes", + Diff::Return { .. } => "return", + Diff::Parameters { .. } => "parameters", + Diff::ParameterAttributes { .. } => "parameter attributes", + } + } +} + +fn compare(c: &Sig, r: &Sig, arch: &str, slot: usize) -> (Vec, Vec) { + let (mut findings, mut notes) = (Vec::new(), Vec::new()); + let width = slot * 8; + if c.cc != r.cc { + findings.push(Diff::CallingConvention { clang: c.cc.clone(), rustc: r.cc.clone() }); + } + if c.ret_attrs != r.ret_attrs { + // x86's psABI leaves the bits above a small integer return undefined (#142389); only + // bool's bits 1-7 must be zero (#163911). + if (arch == "x86_64" || arch == "x86") && c.ret != ["i1"] { + notes.push(format!("return attributes {:?} vs {:?} (x86 psABI: upper bits undefined)", c.ret_attrs, r.ret_attrs)); + } else { + findings.push(Diff::ReturnAttributes { clang: c.ret_attrs.clone(), rustc: r.ret_attrs.clone(), ret: c.ret.clone() }); + } + } + if units(&c.ret, arch, true, width) != units(&r.ret, arch, true, width) { + findings.push(Diff::Return { clang: c.ret.clone(), rustc: r.ret.clone() }); + } else if c.ret != r.ret { + notes.push(format!("return types {:?} vs {:?}", c.ret, r.ret)); + } + // ARM and 64-bit PowerPC split a byval aggregate between registers and stack as they do an + // array argument of the same size: both are "an aggregate". + let agg = |p: &Param| -> Param { + if (arch == "arm" || arch == "powerpc64") && (p.attrs.iter().any(|a| a == "byval") || INT_ARRAY.is_match(&p.ty)) { + Param { ty: "agg".into(), attrs: vec![], noundef: false } + } else { + p.clone() + } + }; + let cp: Vec = c.params.iter().map(agg).collect(); + let rp: Vec = r.params.iter().map(agg).collect(); + let ct: Vec = cp.iter().map(|p| p.ty.clone()).collect(); + let rt: Vec = rp.iter().map(|p| p.ty.clone()).collect(); + if units(&ct, arch, false, width) != units(&rt, arch, false, width) { + // i386 passes every argument on the stack: an expanded struct and a byval copy of it are + // the same bytes. + if arch == "x86" && cp.iter().chain(&rp).any(|p| p.attrs.iter().any(|a| a == "byval")) { + notes.push(format!("parameters {ct:?} vs {rt:?} (x86: same stack bytes)")); + } else { + findings.push(Diff::Parameters { clang: ct, rustc: rt }); + } + } else if ct != rt { + notes.push(format!("parameter types {ct:?} vs {rt:?}")); + } else { + for (i, (a, b)) in cp.iter().zip(&rp).enumerate() { + if a.attrs != b.attrs { + let aligns: Vec = a.attrs.iter().chain(&b.attrs).filter_map(|x| x.strip_prefix("align ")).filter_map(|x| x.parse().ok()).collect(); + let rest_c: Vec<&String> = a.attrs.iter().filter(|x| !x.starts_with("align ")).collect(); + let rest_r: Vec<&String> = b.attrs.iter().filter(|x| !x.starts_with("align ")).collect(); + let only_byval = rest_c.iter().chain(&rest_r).all(|x| x.as_str() == "byval"); + let ext_only_rust = rest_c.is_empty() && rest_r.iter().all(|x| x.as_str() == "zeroext" || x.as_str() == "signext"); + if rest_c == rest_r && !aligns.is_empty() && aligns.iter().max().copied().unwrap_or(0) as usize <= slot { + notes.push(format!("parameter {i} alignment within a stack slot")); + } else if (arch == "wasm32" || arch == "wasm64") && only_byval { + notes.push(format!("parameter {i}: wasm byval is a pointer to a copy")); + } else if arch == "x86" && only_byval { + notes.push(format!("parameter {i}: x86 same stack bytes")); + } else if arch == "x86_64" && ext_only_rust { + // Win64: rustc extends, clang does not; the callee re-extends either way. + notes.push(format!("parameter {i}: win64 extension not relied on")); + } else { + findings.push(Diff::ParameterAttributes { index: i, clang: a.attrs.clone(), rustc: b.attrs.clone(), ty: a.ty.clone() }); + } + } + if a.noundef != b.noundef { + notes.push(format!("parameter {i} noundef")); + } + } + } + (findings, notes) +} + +/// A difference that is a bug already recorded, or one this host cannot decide: its label. +fn label(arch: &str, target: &str, d: &Diff) -> Option<&'static str> { + let fp = |v: &[String]| v.iter().filter(|t| *t == "float" || *t == "double").count(); + match d { + Diff::ReturnAttributes { clang, .. } if arch == "x86_64" && clang == &["zeroext"] => Some("rust-lang/rust#163911"), + Diff::ParameterAttributes { clang, rustc, .. } + if matches!(arch, "riscv64" | "riscv32" | "loongarch64") && rustc.is_empty() && (clang == &["signext"] || clang == &["zeroext"]) => + { + Some("finding 19 (docs/hunt.md)") + } + Diff::Parameters { clang, rustc } | Diff::Return { clang, rustc } if matches!(arch, "riscv64" | "riscv32" | "loongarch64") && fp(rustc) > fp(clang) => { + Some("finding 20 (docs/hunt.md)") + } + // clang returns an 8-byte struct with a 3-byte array field indirectly; rustc and MSVC's + // documentation in edx:eax. Needs MSVC. + Diff::Return { rustc, .. } if arch == "x86" && target.ends_with("windows-msvc") && rustc != &["void"] => { + Some("i686 msvc small-struct return (needs MSVC)") + } + Diff::Parameters { clang, rustc } if arch == "x86" && target.ends_with("windows-msvc") && clang.first().map(String::as_str) == Some("ptr") && clang[1..] == rustc[..] => { + Some("i686 msvc small-struct return (needs MSVC)") + } + // clang marks a float from a single-member aggregate inreg; needs a run on the target. + Diff::ParameterAttributes { clang, rustc, ty, .. } if arch == "powerpc64" && clang == &["inreg"] && rustc.is_empty() && (ty == "float" || ty == "double") => { + Some("ppc64 inreg float (needs a run)") + } + _ => None, + } +} + +// ---- running ---- + +#[derive(Deserialize, Default)] +#[serde(rename_all = "kebab-case")] +struct Spec { + #[serde(default)] + llvm_target: String, + #[serde(default)] + arch: String, + #[serde(default)] + target_pointer_width: serde_json::Value, + #[serde(default)] + c_int_width: serde_json::Value, + #[serde(default)] + cpu: String, + #[serde(default)] + features: String, + #[serde(default)] + llvm_abiname: String, + #[serde(default)] + llvm_floatabi: String, +} + +fn num(v: &serde_json::Value, default: usize) -> usize { + v.as_u64().map(|x| x as usize).or_else(|| v.as_str().and_then(|s| s.parse().ok())).unwrap_or(default) +} + +#[derive(Serialize, Default)] +struct TargetResult { + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + compared: usize, + findings: BTreeMap>, + labelled: BTreeMap, + notes: usize, +} + +fn run_cmd(cmd: &mut Command) -> Result<(), String> { + match cmd.env("RUSTC_BOOTSTRAP", "1").output() { + Ok(o) if o.status.success() => Ok(()), + Ok(o) => Err(String::from_utf8_lossy(&o.stderr).chars().rev().take(1000).collect::().chars().rev().collect()), + Err(e) => Err(e.to_string()), + } +} + +fn one_target(args: &Args, target: &str) -> TargetResult { + let skip = |why: String| TargetResult { skip: Some(why), ..Default::default() }; + let spec_out = Command::new(&args.rustc).args(["--print", "target-spec-json", "-Zunstable-options", "--target", target]).env("RUSTC_BOOTSTRAP", "1").output(); + let Ok(spec_out) = spec_out else { return skip("no spec".into()) }; + let Ok(spec) = serde_json::from_slice::(&spec_out.stdout) else { return skip("no spec".into()) }; + let width = num(&spec.target_pointer_width, 64); + if width < 32 || num(&spec.c_int_width, 32) != 32 { + return skip("16-bit int".into()); + } + let d = args.work.join(target); + let _ = std::fs::create_dir_all(&d); + let (csrc, rsrc, fns) = program(width == 64 && spec.arch != "sparc64", args.seed, args.count); + let _ = std::fs::write(d.join("a.c"), &csrc); + let _ = std::fs::write(d.join("a.rs"), &rsrc); + let base = |c: &mut Command| { + c.args(["--target", target, "-Zunstable-options", "--edition", "2021", "-Cpanic=abort", "--out-dir"]).arg(&d); + }; + let mut mc = Command::new(&args.rustc); + base(&mut mc); + mc.args(["--crate-type", "rlib", "--crate-name", "minicore", "-Awarnings"]).arg(args.rust.join("tests/auxiliary/minicore.rs")); + if let Err(e) = run_cmd(&mut mc) { + return skip(format!("minicore does not build: {e}")); + } + let mut rc = Command::new(&args.rustc); + base(&mut rc); + rc.args(["--emit=llvm-ir", "-Copt-level=0", "--extern"]).arg(format!("minicore={}", d.join("libminicore.rlib").display())).arg("-o").arg(d.join("r.ll")).arg(d.join("a.rs")); + if let Err(e) = run_cmd(&mut rc) { + return skip(format!("rust side does not build: {e}")); + } + let mut cc = Command::new(&args.clang); + cc.arg(format!("--target={}", spec.llvm_target)).args(["-ffreestanding", "-S", "-emit-llvm", "-O0", "-Wno-everything"]); + // The target's CPU and features decide parts of the ABI in clang too (soft-float, SSE). + if !spec.cpu.is_empty() && spec.cpu != "generic" { + cc.args(["-Xclang", "-target-cpu", "-Xclang", &spec.cpu]); + } + for f in spec.features.split(',').filter(|f| !f.is_empty()) { + cc.args(["-Xclang", "-target-feature", "-Xclang", f]); + } + if !spec.llvm_abiname.is_empty() { + cc.arg(format!("-mabi={}", spec.llvm_abiname)); + } + if spec.llvm_floatabi == "hard" { + cc.arg("-mfloat-abi=hard"); + } + cc.arg("-o").arg(d.join("c.ll")).arg(d.join("a.c")); + if let Err(e) = run_cmd(&mut cc) { + return skip(format!("clang does not build: {e}")); + } + let cs = signatures(&std::fs::read_to_string(d.join("c.ll")).unwrap_or_default()); + let rsig = signatures(&std::fs::read_to_string(d.join("r.ll")).unwrap_or_default()); + let mut res = TargetResult { compared: fns.len(), ..Default::default() }; + for name in &fns { + let (Some(c), Some(r)) = (cs.get(name), rsig.get(name)) else { continue }; + let (f, n) = compare(c, r, &spec.arch, width / 8); + res.notes += n.len(); + let mut unlabelled = Vec::new(); + for diff in f { + match label(&spec.arch, target, &diff) { + Some(l) => *res.labelled.entry(l.to_owned()).or_default() += 1, + None => unlabelled.push(diff), + } + } + if !unlabelled.is_empty() { + res.findings.insert(name.clone(), unlabelled); + } + } + res +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let targets: Vec = if let Some(t) = &args.targets { + t.split(',').map(str::to_owned).collect() + } else if args.all { + let out = Command::new(&args.rustc).args(["--print", "target-list"]).output()?; + String::from_utf8_lossy(&out.stdout).split_whitespace().map(str::to_owned).collect() + } else { + MAIN.iter().map(|s| s.to_string()).collect() + }; + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let results: BTreeMap = pool.install(|| targets.par_iter().map(|t| (t.clone(), one_target(&args, t))).collect()); + std::fs::write(args.work.join("results.json"), serde_json::to_string_pretty(&results)?)?; + let mut labelled: BTreeMap<&str, usize> = BTreeMap::new(); + let mut kinds: BTreeMap<&str, BTreeMap<&str, usize>> = BTreeMap::new(); + let (mut compared, mut skipped) = (0, BTreeMap::::new()); + for (t, r) in &results { + if let Some(s) = &r.skip { + *skipped.entry(s.split(':').next().unwrap_or("").to_owned()).or_default() += 1; + continue; + } + compared += 1; + for (l, n) in &r.labelled { + *labelled.entry(l.as_str()).or_default() += n; + } + for diffs in r.findings.values() { + for d in diffs { + *kinds.entry(d.label()).or_default().entry(t.as_str()).or_default() += 1; + } + } + } + println!("{compared} targets compared; skipped: {skipped:?}"); + if !labelled.is_empty() { + println!("known: {labelled:?}"); + } + for (kind, per) in &kinds { + let total: usize = per.values().sum(); + let mut top: Vec<(&&str, &usize)> = per.iter().collect(); + top.sort_by(|a, b| b.1.cmp(a.1)); + println!("{kind}: {total} in {} targets: {}", per.len(), top.iter().take(8).map(|(t, n)| format!("{t} {n}")).collect::>().join(", ")); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/release_diff.rs b/crates/mirth-lab/src/tools/release_diff.rs new file mode 100644 index 0000000..b56ae9e --- /dev/null +++ b/crates/mirth-lab/src/tools/release_diff.rs @@ -0,0 +1,158 @@ +//! Release-to-release: code that one toolchain accepts the next must accept too, in comparable +//! time. +//! +//! `cargo check --locked --offline --workspace` of each repository in a corpus of real crates +//! with an older and a newer toolchain (dependencies fetched first), each in its own target +//! directory, removed afterwards. Findings: a regression (old passes, new fails), an ICE, or the +//! new toolchain taking more than --slower times as long. A regression whose failing crate +//! enables unstable features (`#![feature]`, often only when it detects a nightly) is noted, not +//! reported. + +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::{Duration, Instant}; + +use mirth_lab::rustc::{Exit, error_codes, is_ice, run_command}; +use rayon::prelude::*; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// A directory of repositories (each with a Cargo.toml). + #[arg(long)] + corpus: PathBuf, + #[arg(long)] + old: String, + #[arg(long)] + new: String, + #[arg(long)] + work: PathBuf, + #[arg(long)] + only: Option, + #[arg(long, default_value_t = 2)] + jobs: usize, + #[arg(long, default_value_t = 1.5)] + slower: f64, + #[arg(long, default_value_t = 1800)] + timeout: u64, +} + +static ERROR_LINE: LazyLock = LazyLock::new(|| Regex::new(r"\berror(\[E\d+\])?:").unwrap()); +static CRATE_ROOT: LazyLock = LazyLock::new(|| Regex::new(r"(/\S*?/registry/src/[^/]+/[^/]+|/\S+?)/src/").unwrap()); +static FEATURE: LazyLock = LazyLock::new(|| Regex::new(r"#!\[(cfg_attr\([^\]]*)?feature\(").unwrap()); + +#[derive(Serialize)] +struct Check { + exit: Exit, + seconds: f64, + ice: bool, + codes: Vec, + first: String, + tail: String, +} + +#[derive(Serialize)] +struct Rec { + repo: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + #[serde(skip_serializing_if = "Option::is_none")] + old: Option, + #[serde(skip_serializing_if = "Option::is_none")] + new: Option, + found: Vec, + notes: Vec, +} + +fn check(args: &Args, repo: &Path, toolchain: &str) -> Check { + let name = repo.file_name().unwrap().to_string_lossy(); + let target = args.work.join("target").join(format!("{name}-{toolchain}")); + let _ = std::fs::remove_dir_all(&target); + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{toolchain}")) + .args(["check", "--locked", "--offline", "--workspace", "--message-format=short"]) + .current_dir(repo) + .env("CARGO_TARGET_DIR", &target) + .env("CARGO_TERM_COLOR", "never") + .env("CARGO_INCREMENTAL", "0") + .env("RUSTFLAGS", "--cap-lints=warn") + .env_remove("RUSTC_WRAPPER"); + let start = Instant::now(); + let done = run_command(cmd, Duration::from_secs(args.timeout)); + let seconds = (start.elapsed().as_secs_f64() * 10.0).round() / 10.0; + let _ = std::fs::remove_dir_all(&target); + let (exit, err) = match done { + Ok(d) => (d.exit.clone(), d.stderr_text()), + Err(e) => (Exit::Code(-1), e.to_string()), + }; + let first = err.lines().find(|l| ERROR_LINE.is_match(l)).unwrap_or("").chars().take(300).collect(); + let tail = if exit == Exit::Code(0) { String::new() } else { err.chars().rev().take(3000).collect::().chars().rev().collect() }; + Check { exit, seconds, ice: is_ice(&err), codes: error_codes(&err), first, tail } +} + +/// The crate a first error points into, if its source enables `#![feature(...)]`. +fn uses_unstable(first_error: &str) -> Option { + let root = PathBuf::from(&CRATE_ROOT.captures(first_error)?[1]); + ["lib.rs", "main.rs"].iter().find_map(|f| { + let text = std::fs::read_to_string(root.join("src").join(f)).ok()?; + FEATURE.is_match(&text).then(|| root.file_name().unwrap().to_string_lossy().into_owned()) + }) +} + +fn one(args: &Args, repo: &Path) -> Rec { + let name = repo.file_name().unwrap().to_string_lossy().into_owned(); + let mut fetch = Command::new("cargo"); + fetch.arg(format!("+{}", args.new)).args(["fetch", "--locked"]).current_dir(repo); + match run_command(fetch, Duration::from_secs(1800)) { + Ok(d) if d.success() => {} + other => { + let why = other.map(|d| d.stderr_text()).unwrap_or_else(|e| e.to_string()); + return Rec { repo: name, skip: Some(format!("fetch failed: {}", why.chars().rev().take(300).collect::().chars().rev().collect::())), old: None, new: None, found: vec![], notes: vec![] }; + } + } + let old = check(args, repo, &args.old); + let new = check(args, repo, &args.new); + let (mut found, mut notes) = (Vec::new(), Vec::new()); + let ok = |c: &Check| c.exit == Exit::Code(0); + if ok(&old) && !ok(&new) { + match uses_unstable(&new.first) { + Some(krate) if !new.ice => notes.push(format!("regression in a crate using unstable features ({krate})")), + _ => found.push(if new.ice { "ice".into() } else { "regression".into() }), + } + } else if !ok(&old) && ok(&new) { + notes.push("fixed".into()); + } else if ok(&old) && ok(&new) && old.seconds > 5.0 && new.seconds > args.slower * old.seconds { + found.push(format!("slower: {}s -> {}s", old.seconds, new.seconds)); + } + if new.ice && !found.iter().any(|f| f == "ice") { + found.push("ice".into()); + } + Rec { repo: name, skip: None, old: Some(old), new: Some(new), found, notes } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let mut repos: Vec = std::fs::read_dir(&args.corpus)? + .flatten() + .map(|e| e.path()) + .filter(|p| p.join("Cargo.toml").exists() && args.only.as_ref().is_none_or(|o| p.to_string_lossy().contains(o.as_str()))) + .collect(); + repos.sort(); + println!("{} repositories, {} -> {}", repos.len(), args.old, args.new); + let out = std::sync::Mutex::new(std::fs::OpenOptions::new().create(true).append(true).open(args.work.join("results.jsonl"))?); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + pool.install(|| { + repos.par_iter().for_each(|repo| { + let rec = one(&args, repo); + if let Ok(mut f) = out.lock() { + use std::io::Write; + let _ = writeln!(f, "{}", serde_json::to_string(&rec).unwrap_or_default()); + } + let tag = rec.skip.clone().unwrap_or_else(|| if rec.found.is_empty() { rec.notes.join(", ") } else { rec.found.join(", ") }); + println!("{:45} {}", rec.repo, if tag.is_empty() { "same".into() } else { tag }); + }) + }); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/scale_check.rs b/crates/mirth-lab/src/tools/scale_check.rs new file mode 100644 index 0000000..50f840b --- /dev/null +++ b/crates/mirth-lab/src/tools/scale_check.rs @@ -0,0 +1,282 @@ +//! Scaling and budgets: compile time, memory, future sizes and stack frames must grow at most +//! about linearly with the size of a program of a fixed shape. +//! +//! Each generator writes a program of size N for N in a doubling series. For each N: the +//! compiler's user CPU time and peak memory (wait4's rusage), the largest `sub $X, %rsp` in the +//! assembly, and for some shapes a size the program reports (`size_of_val` of a future). The +//! growth exponent k (value ~ N^k) is the log-log slope over the three largest sizes. Findings: +//! k above --max-exponent for time or memory, above 1.3 for a frame or a size, or a timeout. + +use std::fmt::Write as _; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode, Stdio}; +use std::sync::LazyLock; +use std::time::{Duration, Instant}; + +use rayon::prelude::*; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[arg(long)] + work: PathBuf, + /// Comma-separated shapes (default: all). + #[arg(long)] + only: Option, + #[arg(long, default_value = "0,2")] + opt: String, + #[arg(long, default_value = "50,100,200,400,800")] + sizes: String, + #[arg(long, default_value_t = 300)] + timeout: u64, + #[arg(long, default_value_t = 1.6)] + max_exponent: f64, + #[arg(long, default_value_t = 4)] + jobs: usize, +} + +fn g_fields(n: usize) -> String { + let mut f = String::new(); + for i in 0..n { + let _ = writeln!(f, " pub f{i}: u{},", 8 << (i % 4)); + } + format!("#[derive(Debug, Clone, PartialEq, Eq, Hash, Default, PartialOrd, Ord)]\npub struct S {{\n{f}}}\nfn main() {{ let s = S::default(); println!(\"{{}}\", format!(\"{{:?}}\", s.clone()).len()); }}\n") +} + +fn g_enum(n: usize) -> String { + let variants: String = (0..n).map(|i| format!(" V{i}(u32),\n")).collect(); + let arms: String = (0..n).map(|i| format!(" E::V{i}(x) => x + {i},\n")).collect(); + format!("#[derive(Debug, Clone, PartialEq)]\npub enum E {{\n{variants}}}\npub fn f(e: &E) -> u32 {{\n match *e {{\n{arms} }}\n}}\nfn main() {{ println!(\"{{}}\", f(&E::V0(1))); }}\n") +} + +fn g_nested_generic(n: usize) -> String { + let ty = (0..n).fold("u8".to_owned(), |t, _| format!("W<{t}>")); + format!("#[derive(Clone, Debug, Default)] pub struct W(T);\npub trait T {{ fn t(&self) -> usize; }}\nimpl T for u8 {{ fn t(&self) -> usize {{ 1 }} }}\nimpl T for W {{ fn t(&self) -> usize {{ self.0.t() + 1 }} }}\nfn main() {{ let v: {ty} = Default::default(); println!(\"{{}}\", v.t()); }}\n") +} + +fn g_iter_chain(n: usize) -> String { + let chain: String = (0..n).map(|i| format!(".map(|x| x.wrapping_add({i}))")).collect(); + format!("fn main() {{ let s: u64 = (0u64..10){chain}.sum(); println!(\"{{}}\", s); }}\n") +} + +fn g_async_forward(n: usize) -> String { + let mut fns = vec!["async fn f0(x: [u8; 64]) -> u8 { x[0] }".to_owned()]; + fns.extend((1..n).map(|i| format!("async fn f{i}(x: [u8; 64]) -> u8 {{ f{}(x).await }}", i - 1))); + format!("{}\nfn main() {{ let fut = f{}([1; 64]); println!(\"SIZE {{}}\", std::mem::size_of_val(&fut)); }}\n", fns.join("\n"), n - 1) +} + +fn g_seq_calls(n: usize) -> String { + let calls: String = (0..n).map(|i| format!(" let a{i} = big({i}); acc ^= a{i}[{}];\n", i % 512)).collect(); + format!("#[inline(never)] fn big(x: u64) -> [u64; 512] {{ [x; 512] }}\n#[inline(never)] pub fn many() -> u64 {{\n let mut acc = 0u64;\n{calls} acc\n}}\nfn main() {{ println!(\"{{}}\", many()); }}\n") +} + +fn g_trait_impls(n: usize) -> String { + let impls: String = (0..n).map(|i| format!("pub struct S{i}; impl Tr for S{i} {{ fn v(&self) -> u32 {{ {i} }} }}\n")).collect(); + let uses = (0..n).map(|i| format!("S{i}.v()")).collect::>().join(" + "); + format!("pub trait Tr {{ fn v(&self) -> u32; }}\n{impls}fn main() {{ println!(\"{{}}\", {uses}); }}\n") +} + +fn g_nested_expr(n: usize) -> String { + let expr = (0..n).fold("1u64".to_owned(), |e, i| format!("({e} + {})", i % 7)); + format!("fn main() {{ let x = std::hint::black_box({expr}); println!(\"{{}}\", x); }}\n") +} + +#[derive(Clone, Copy, PartialEq)] +enum Extra { + None, + RunSize, + Frame, +} + +/// (name, generator, what else to measure, largest meaningful N) +const SHAPES: &[(&str, fn(usize) -> String, Extra, usize)] = &[ + ("fields", g_fields, Extra::None, usize::MAX), + ("enum", g_enum, Extra::None, usize::MAX), + ("nested-generic", g_nested_generic, Extra::None, 120), + ("iter-chain", g_iter_chain, Extra::None, usize::MAX), + ("async-forward", g_async_forward, Extra::RunSize, 200), + ("seq-calls", g_seq_calls, Extra::Frame, usize::MAX), + ("trait-impls", g_trait_impls, Extra::None, usize::MAX), + ("nested-expr", g_nested_expr, Extra::None, 400), +]; + +#[derive(Serialize, Default, Clone)] +struct Row { + n: usize, + #[serde(skip_serializing_if = "Option::is_none")] + user: Option, + #[serde(skip_serializing_if = "Option::is_none")] + rss_kb: Option, + max_frame: u64, + #[serde(skip_serializing_if = "Option::is_none")] + run_size: Option, + timeout: bool, + #[serde(skip_serializing_if = "String::is_empty")] + error: String, +} + +/// Compile in `dir` with rusage from wait4; None user time on timeout. +fn compile_measured(args: &Args, dir: &Path, opt: u32) -> Row { + let mut child = match Command::new(&args.rustc) + .args(["m.rs", "--edition", "2021", &format!("-Copt-level={opt}"), "-o", "m", "--emit=link,asm"]) + .current_dir(dir) + .stdout(Stdio::null()) + .stderr(Stdio::piped()) + .spawn() + { + Ok(c) => c, + Err(e) => return Row { error: e.to_string(), ..Default::default() }, + }; + let pid = child.id() as libc::pid_t; + let deadline = Instant::now() + Duration::from_secs(args.timeout); + let mut status: libc::c_int = 0; + // SAFETY: rusage is plain data; wait4 fills it for the child we spawned. + let mut usage: libc::rusage = unsafe { std::mem::zeroed() }; + loop { + // SAFETY: waiting on our own child; status and usage outlive the call. + let r = unsafe { libc::wait4(pid, &mut status, libc::WNOHANG, &mut usage) }; + if r == pid { + break; + } + if Instant::now() > deadline { + let _ = child.kill(); + // SAFETY: reap the killed child. + unsafe { libc::wait4(pid, &mut status, 0, &mut usage) }; + return Row { timeout: true, ..Default::default() }; + } + std::thread::sleep(Duration::from_millis(50)); + } + let mut err = String::new(); + if let Some(mut e) = child.stderr.take() { + use std::io::Read; + let _ = e.read_to_string(&mut err); + } + if !(libc::WIFEXITED(status) && libc::WEXITSTATUS(status) == 0) { + return Row { error: err.chars().rev().take(500).collect::().chars().rev().collect(), ..Default::default() }; + } + let user = usage.ru_utime.tv_sec as f64 + usage.ru_utime.tv_usec as f64 / 1e6; + Row { user: Some(user), rss_kb: Some(usage.ru_maxrss), ..Default::default() } +} + +static SUB_RSP: LazyLock = LazyLock::new(|| Regex::new(r"subq?\s+\$(0x[0-9a-f]+|\d+),\s*%rsp").unwrap()); +static SIZE: LazyLock = LazyLock::new(|| Regex::new(r"SIZE (\d+)").unwrap()); + +fn measure(args: &Args, source: &str, opt: u32) -> Row { + let dir = tempfile::tempdir_in(&args.work).expect("scratch"); + let _ = std::fs::write(dir.path().join("m.rs"), source); + let mut row = compile_measured(args, dir.path(), opt); + if row.user.is_none() { + return row; + } + let asm = std::fs::read_to_string(dir.path().join("m.s")).unwrap_or_default(); + row.max_frame = SUB_RSP + .captures_iter(&asm) + .filter_map(|c| { + let v = &c[1]; + if let Some(h) = v.strip_prefix("0x") { u64::from_str_radix(h, 16).ok() } else { v.parse().ok() } + }) + .max() + .unwrap_or(0); + if let Ok(out) = Command::new(dir.path().join("m")).output() { + row.run_size = SIZE.captures(&String::from_utf8_lossy(&out.stdout)).and_then(|c| c[1].parse().ok()); + } + row +} + +/// The log-log slope over the three largest positive points. +fn exponent(points: &[(usize, f64)]) -> Option { + let pts: Vec<(f64, f64)> = points.iter().filter(|(_, v)| *v > 0.0).map(|&(n, v)| ((n as f64).ln(), v.ln())).collect(); + if pts.len() < 3 { + return None; + } + let pts = &pts[pts.len() - 3..]; + let mx = pts.iter().map(|p| p.0).sum::() / 3.0; + let my = pts.iter().map(|p| p.1).sum::() / 3.0; + let den: f64 = pts.iter().map(|p| (p.0 - mx).powi(2)).sum(); + (den > 0.0).then(|| pts.iter().map(|p| (p.0 - mx) * (p.1 - my)).sum::() / den) +} + +#[derive(Serialize)] +struct Res { + shape: String, + opt: u32, + rows: Vec, + k_time: Option, + k_rss: Option, + k_frame: Option, + k_size: Option, + found: Vec, +} + +fn one(args: &Args, sizes: &[usize], shape: &(&str, fn(usize) -> String, Extra, usize), opt: u32) -> Res { + let (name, gen_fn, extra, cap) = *shape; + let mut rows = Vec::new(); + for &n in sizes.iter().filter(|&&n| n <= cap) { + let mut r = measure(args, &gen_fn(n), opt); + r.n = n; + let stop = r.timeout || !r.error.is_empty(); + rows.push(r); + if stop { + break; + } + } + let ok: Vec<&Row> = rows.iter().filter(|r| r.user.is_some()).collect(); + let first_user = ok.first().and_then(|r| r.user).unwrap_or(0.0); + let k_time = exponent(&ok.iter().map(|r| (r.n, (r.user.unwrap() - first_user * 0.5).max(1e-3))).collect::>()); + let k_rss = exponent(&ok.iter().map(|r| (r.n, r.rss_kb.unwrap_or(0) as f64)).collect::>()); + let k_frame = (extra == Extra::Frame).then(|| exponent(&ok.iter().map(|r| (r.n, r.max_frame as f64)).collect::>())).flatten(); + let k_size = (extra == Extra::RunSize).then(|| exponent(&ok.iter().map(|r| (r.n, r.run_size.unwrap_or(0) as f64)).collect::>())).flatten(); + let mut found = Vec::new(); + if let Some(last) = rows.last() { + if last.timeout { + found.push(format!("timeout at N={}", last.n)); + } + if !last.error.is_empty() { + found.push(format!("error at N={}: {}", last.n, last.error.chars().rev().take(200).collect::().chars().rev().collect::())); + } + } + for (label, k, limit) in [("k_time", k_time, args.max_exponent), ("k_rss", k_rss, args.max_exponent), ("k_frame", k_frame, 1.3), ("k_size", k_size, 1.3)] { + if let Some(k) = k + && k > limit + { + found.push(format!("{label} = {k:.2}")); + } + } + Res { shape: name.into(), opt, rows, k_time, k_rss, k_frame, k_size, found } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let sizes: Vec = args.sizes.split(',').filter_map(|s| s.parse().ok()).collect(); + let opts: Vec = args.opt.split(',').filter_map(|s| s.parse().ok()).collect(); + let wanted: Option> = args.only.as_deref().map(|o| o.split(',').collect()); + let jobs: Vec<(&(&str, fn(usize) -> String, Extra, usize), u32)> = SHAPES + .iter() + .filter(|s| wanted.as_ref().is_none_or(|w| w.contains(&s.0))) + .flat_map(|s| opts.iter().map(move |&o| (s, o))) + .collect(); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let results: Vec = pool.install(|| jobs.par_iter().map(|(s, o)| one(&args, &sizes, s, *o)).collect()); + std::fs::write(args.work.join("results.json"), serde_json::to_string_pretty(&results)?)?; + for r in &results { + let last = r.rows.iter().rev().find(|x| x.user.is_some()); + let ks: Vec = [("k_time", r.k_time), ("k_rss", r.k_rss), ("k_frame", r.k_frame), ("k_size", r.k_size)] + .iter() + .filter_map(|(l, k)| k.map(|k| format!("{l}={k:.2}"))) + .collect(); + println!( + "{:15} O{} N<={} user {:.1}s rss {}MB {} {}", + r.shape, + r.opt, + last.map_or(0, |x| x.n), + last.and_then(|x| x.user).unwrap_or(0.0), + last.and_then(|x| x.rss_kb).unwrap_or(0) / 1024, + ks.join(" "), + r.found.join("; ") + ); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/xlink.rs b/crates/mirth-lab/src/tools/xlink.rs new file mode 100644 index 0000000..526c9ef --- /dev/null +++ b/crates/mirth-lab/src/tools/xlink.rs @@ -0,0 +1,173 @@ +//! Cross-target build and link: every target must build `core` and `alloc` and link a program +//! with its documented linker, with no undefined symbols. +//! +//! For each target, builds rustc/xlink-probe (a `no_std` program needing compiler-builtins: 128-bit +//! integers, float conversions and math, float formatting, large copies, atomics) with +//! `cargo -Zbuild-std=core,alloc` and links it; targets whose default linker is a C compiler +//! driver link with `rust-lld` in the flavor their spec names. Results: ok, link-undefined (the +//! finding: lld reports undefined symbols), link, env (a library or startup file of the target's +//! C sysroot missing here), build, ice, skipped. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::Duration; + +use mirth_lab::rustc::{Exit, run_command}; +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + toolchain: String, + #[arg(long)] + work: PathBuf, + /// Comma-separated targets (default: every target rustc knows). + #[arg(long)] + targets: Option, + #[arg(long, default_value_t = 6)] + jobs: usize, + /// The probe crate (default: rustc/xlink-probe in the mirth checkout). + #[arg(long)] + probe: Option, +} + +/// Targets that need more than a target name (a CPU, a linker that is not lld). +static SKIP: LazyLock = LazyLock::new(|| Regex::new(r"^(amdgcn|nvptx|bpf|spirv)|avr-none").unwrap()); +static UNDEFINED: LazyLock = LazyLock::new(|| Regex::new(r"undefined symbol: (\S+)").unwrap()); +static ENV_MISSING: LazyLock = LazyLock::new(|| Regex::new(r"unable to find library|cannot open crt|cannot open .*\.o\b|No such file").unwrap()); + +/// The parts of a target spec the link options depend on. +#[derive(Deserialize, Default)] +#[serde(rename_all = "kebab-case")] +struct Spec { + #[serde(default)] + linker_flavor: String, + #[serde(default)] + linker: String, + #[serde(default)] + is_like_wasm: bool, + #[serde(default)] + is_like_msvc: bool, + #[serde(default)] + is_like_darwin: bool, +} + +#[derive(Serialize)] +struct Res { + result: String, + #[serde(skip_serializing_if = "Vec::is_empty")] + flags: Vec, + #[serde(skip_serializing_if = "Vec::is_empty")] + undefined: Vec, + #[serde(skip_serializing_if = "String::is_empty")] + first: String, + #[serde(skip_serializing_if = "String::is_empty")] + tail: String, +} + +fn link_flags(spec: &Spec) -> Vec { + let f = spec.linker_flavor.as_str(); + let own_lld = spec.linker.contains("lld") || f.ends_with("-lld") || f.starts_with("wasm-lld"); + let mut flags: Vec<&str> = Vec::new(); + if spec.is_like_wasm || f.starts_with("wasm") { + flags.extend(["-Clink-arg=--no-entry", "-Clink-arg=--export=probe_entry"]); + } else if spec.is_like_msvc || f.starts_with("msvc") { + if !own_lld { + flags.extend(["-Clinker=rust-lld", "-Clinker-flavor=lld-link"]); + } + flags.extend(["-Clink-arg=/ENTRY:probe_entry", "-Clink-arg=/NODEFAULTLIB"]); + } else if spec.is_like_darwin || f.starts_with("darwin") { + if !own_lld { + flags.extend(["-Clinker=rust-lld", "-Clinker-flavor=ld64.lld"]); + } + flags.extend(["-Clink-arg=-e", "-Clink-arg=_probe_entry", "-Clink-arg=-undefined", "-Clink-arg=dynamic_lookup"]); + } else { + if !own_lld { + flags.extend(["-Clinker=rust-lld", "-Clinker-flavor=ld.lld"]); + } + flags.push("-Clink-arg=--entry=probe_entry"); + } + flags.into_iter().map(str::to_owned).collect() +} + +fn rustc_out(toolchain: &str, a: &[&str]) -> Option { + let out = Command::new("rustc").arg(format!("+{toolchain}")).args(a).env("RUSTC_BOOTSTRAP", "1").output().ok()?; + out.status.success().then(|| String::from_utf8_lossy(&out.stdout).into_owned()) +} + +fn one(args: &Args, probe: &Path, target: &str) -> Res { + let skipped = |why: &str| Res { result: "skipped".into(), flags: vec![], undefined: vec![], first: why.into(), tail: String::new() }; + if SKIP.is_match(target) { + return skipped(""); + } + let Some(spec_json) = rustc_out(&args.toolchain, &["--print", "target-spec-json", "-Zunstable-options", "--target", target]) else { + return skipped("no spec"); + }; + let spec: Spec = serde_json::from_str(&spec_json).unwrap_or_default(); + let flags = link_flags(&spec); + let tdir = args.work.join("target").join(target); + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{}", args.toolchain)) + .args(["build", "--release", "-Zbuild-std=core,alloc", "-Zbuild-std-features=compiler-builtins-mem", "--target", target]) + .current_dir(probe) + .env("CARGO_TARGET_DIR", &tdir) + .env("RUSTFLAGS", flags.join(" ")) + .env("CARGO_TERM_COLOR", "never") + .env_remove("RUSTC_WRAPPER"); + let done = run_command(cmd, Duration::from_secs(1500)); + let _ = std::fs::remove_dir_all(&tdir); + let (exit, err) = match done { + Ok(d) => (d.exit.clone(), d.stderr_text()), + Err(e) => (Exit::Code(-1), e.to_string()), + }; + let result = if exit == Exit::Code(0) { + "ok" + } else if err.contains("internal compiler error") || err.contains("panicked at") { + "ice" + } else if err.contains("linking with") || err.contains("lld: error") { + if err.contains("undefined symbol") || err.contains("undefined reference") { + "link-undefined" + } else if ENV_MISSING.is_match(&err) { + "env" + } else { + "link" + } + } else { + "build" + }; + let mut undefined: Vec = UNDEFINED.captures_iter(&err).map(|c| c[1].to_owned()).collect(); + undefined.sort(); + undefined.dedup(); + undefined.truncate(20); + let first = err.lines().find(|l| l.contains("error[") || l.contains("error:")).unwrap_or("").chars().take(300).collect(); + let tail = if result == "ok" { String::new() } else { err.chars().rev().take(2500).collect::().chars().rev().collect() }; + Res { result: result.into(), flags, undefined, first, tail } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let probe = args.probe.clone().unwrap_or_else(|| PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../../rustc/xlink-probe")); + let targets: Vec = match &args.targets { + Some(t) => t.split(',').map(str::to_owned).collect(), + None => rustc_out(&args.toolchain, &["--print", "target-list"]).unwrap_or_default().split_whitespace().map(str::to_owned).collect(), + }; + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let results: BTreeMap = pool.install(|| targets.par_iter().map(|t| (t.clone(), one(&args, &probe, t))).collect()); + std::fs::write(args.work.join("results.json"), serde_json::to_string_pretty(&results)?)?; + let mut counts: BTreeMap<&str, usize> = BTreeMap::new(); + for r in results.values() { + *counts.entry(r.result.as_str()).or_default() += 1; + } + println!("{counts:?}"); + for (t, r) in &results { + if r.result != "ok" && r.result != "skipped" && r.result != "env" { + let what = if r.undefined.is_empty() { r.first.chars().take(120).collect() } else { r.undefined.join(" ") }; + println!("{:15} {t:40} {what}", r.result); + } + } + Ok(ExitCode::SUCCESS) +} From 542bf98edaef7f2ac1c1bfdc280c69992aeec489 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 07:45:57 +0000 Subject: [PATCH 11/23] The Python oracle scripts removed (ported to mirth-lab and validated); docs name the mirth-lab subcommands; how to run them in checks.md Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- docs/checks.md | 43 +- docs/hunt.md | 2 +- docs/hunt/glob-ambiguity-order.md | 2 +- docs/hunt/internal-checks.md | 4 +- docs/hunt/lint-fixes-break-builds.md | 2 +- docs/hunt/promoted-validation-generic.md | 4 +- docs/hunt/release-regressions.md | 2 +- docs/hunt/riscv-float-pointer-struct.md | 2 +- docs/hunt/riscv-stack-arg-extension.md | 2 +- docs/solver-triage.md | 4 +- docs/solver.md | 4 +- rustc/abi-diff.py | 468 ------------------ rustc/crash-diff.py | 93 ---- rustc/diag-check.py | 171 ------- rustc/gate-check.py | 200 -------- rustc/instr-check.py | 162 ------ rustc/miri-diff.py | 168 ------- rustc/opt-diff.py | 185 ------- rustc/release-diff.py | 124 ----- rustc/repro-diff.py | 146 ------ rustc/rewrite-diff.py | 176 ------- rustc/scale-check.py | 194 -------- rustc/solver-diff.py | 127 ----- rustc/suggest-diff.py | 195 -------- rustc/uitest.py | 161 ------ .../rustc-ice-2026-10-10T06_52_34-1073611.txt | 32 ++ .../rustc-ice-2026-10-10T06_54_55-1317662.txt | 32 ++ .../rustc-ice-2026-10-10T06_57_29-1334708.txt | 32 ++ .../rustc-ice-2026-10-10T06_59_46-1343102.txt | 32 ++ .../rustc-ice-2026-10-10T06_59_59-1343572.txt | 32 ++ .../rustc-ice-2026-10-10T07_00_16-1344795.txt | 32 ++ .../rustc-ice-2026-10-10T07_00_31-1345593.txt | 32 ++ .../rustc-ice-2026-10-10T07_02_04-1382261.txt | 32 ++ .../rustc-ice-2026-10-10T07_10_33-1720154.txt | 32 ++ rustc/xlink.py | 123 ----- 35 files changed, 332 insertions(+), 2720 deletions(-) delete mode 100644 rustc/abi-diff.py delete mode 100644 rustc/crash-diff.py delete mode 100644 rustc/diag-check.py delete mode 100644 rustc/gate-check.py delete mode 100644 rustc/instr-check.py delete mode 100644 rustc/miri-diff.py delete mode 100644 rustc/opt-diff.py delete mode 100644 rustc/release-diff.py delete mode 100644 rustc/repro-diff.py delete mode 100644 rustc/rewrite-diff.py delete mode 100644 rustc/scale-check.py delete mode 100644 rustc/solver-diff.py delete mode 100644 rustc/suggest-diff.py delete mode 100644 rustc/uitest.py create mode 100644 rustc/xlink-probe/rustc-ice-2026-10-10T06_52_34-1073611.txt create mode 100644 rustc/xlink-probe/rustc-ice-2026-10-10T06_54_55-1317662.txt create mode 100644 rustc/xlink-probe/rustc-ice-2026-10-10T06_57_29-1334708.txt create mode 100644 rustc/xlink-probe/rustc-ice-2026-10-10T06_59_46-1343102.txt create mode 100644 rustc/xlink-probe/rustc-ice-2026-10-10T06_59_59-1343572.txt create mode 100644 rustc/xlink-probe/rustc-ice-2026-10-10T07_00_16-1344795.txt create mode 100644 rustc/xlink-probe/rustc-ice-2026-10-10T07_00_31-1345593.txt create mode 100644 rustc/xlink-probe/rustc-ice-2026-10-10T07_02_04-1382261.txt create mode 100644 rustc/xlink-probe/rustc-ice-2026-10-10T07_10_33-1720154.txt delete mode 100644 rustc/xlink.py diff --git a/docs/checks.md b/docs/checks.md index a3a362f..785c174 100644 --- a/docs/checks.md +++ b/docs/checks.md @@ -210,7 +210,7 @@ partly covered by checks 1, 9 and 17. UB (#160670, #160669), and diagnostic differences between the solvers (#162919). The check: compile everything with `-Znext-solver=globally` and with the old solver, and with NLL and `-Zpolonius=next`. Compare the verdicts and error codes, and run one-sided acceptances under -Miri. **The cheapest check here: `ui-solver-diff.py` already does half of it** +Miri. **The cheapest check here: `ui-mirth-lab solver-diff` already does half of it** ([solver.md](solver.md)). ### 10. Scaling and budgets (31) @@ -316,7 +316,7 @@ Ordered by bugs caught per unit of effort, and by what mirth already has: 1. **Optimization and pass differential** (45). It extends the option matrix with a run step. Two afternoons. -2. **Solver differential** (32, plus up to 73 ICEs). It is `ui-solver-diff.py` plus Polonius, +2. **Solver differential** (32, plus up to 73 ICEs). It is `ui-mirth-lab solver-diff` plus Polonius, and the next solver as a configuration column for the whole corpus. 3. **Miri differential** (52). It runs Miri over the run-pass tests, then once per MIR pass. 4. **Equivalent rewrites in ui-fuzz** (63). Four rewrites to start with. @@ -351,13 +351,13 @@ over the standalone UI tests at the pin (and real crates for release-to-release) | check | script | swept | result | |---|---|---|---| -| optimization and pass differential | `opt-diff.py` | 3,217 runnable tests × 13 configurations (opt levels, MIR opt levels, LTO, target CPU, Cranelift) | nothing; Cranelift's gaps (tail calls, some linkages and SIMD intrinsics) noted | -| solver differential | `solver-diff.py` | 17,634 tests × old/new solver × NLL/Polonius | the 26 rejections and 3 crashes of [`solver.md`](solver.md); Polonius agrees with NLL everywhere | -| Miri differential | `miri-diff.py` | 3,094 runnable tests at MIR opt levels 0, 2, 4 and natively | nothing; tests asserting unspecified behavior (function pointer equality, ZST addresses) listed | -| equivalent rewrites | `mirth-rewrite` + `rewrite-diff.py` | 18,624 tests × generic-wrap, alias, reorder, unused | findings 25 (generic-wrap) and 28 (reorder) | -| ABI vs clang | `abi-diff.py` | 21 main targets × 10 seeds × 300 random signatures | findings 19 and 20; #163911 reproduced; i686 MSVC small-struct returns and a PowerPC64 `inreg` float undecided | -| internal checks on | `crash-diff.py` + a debug-assertions compiler | 18,624 tests with `-Zvalidate-mir` | findings 21–24 (17 tests) | -| release-to-release | `release-diff.py` | 87 real repositories, nightly-2026-07-18 → 10-06 | findings 26 and 27; `allocative` (unstable features) noted | +| optimization and pass differential | `mirth-lab opt-diff` | 3,217 runnable tests × 13 configurations (opt levels, MIR opt levels, LTO, target CPU, Cranelift) | nothing; Cranelift's gaps (tail calls, some linkages and SIMD intrinsics) noted | +| solver differential | `mirth-lab solver-diff` | 17,634 tests × old/new solver × NLL/Polonius | the 26 rejections and 3 crashes of [`solver.md`](solver.md); Polonius agrees with NLL everywhere | +| Miri differential | `mirth-lab miri-diff` | 3,094 runnable tests at MIR opt levels 0, 2, 4 and natively | nothing; tests asserting unspecified behavior (function pointer equality, ZST addresses) listed | +| equivalent rewrites | `mirth-rewrite` + `mirth-lab rewrite-diff` | 18,624 tests × generic-wrap, alias, reorder, unused | findings 25 (generic-wrap) and 28 (reorder) | +| ABI vs clang | `mirth-lab abi-diff` | 21 main targets × 10 seeds × 300 random signatures | findings 19 and 20; #163911 reproduced; i686 MSVC small-struct returns and a PowerPC64 `inreg` float undecided | +| internal checks on | `mirth-lab crash-diff` + a debug-assertions compiler | 18,624 tests with `-Zvalidate-mir` | findings 21–24 (17 tests) | +| release-to-release | `mirth-lab release-diff` | 87 real repositories, nightly-2026-07-18 → 10-06 | findings 26 and 27; `allocative` (unstable features) noted | Ten new findings (19–28) in [`hunt.md`](hunt.md), none from the checks mirth had before. @@ -365,7 +365,24 @@ Ten new findings (19–28) in [`hunt.md`](hunt.md), none from the checks mirth h | check | script | swept | result | |---|---|---|---| -| suggestions apply (18) | `suggest-diff.py` | 17,945 tests without `run-rustfix`, 7,385 machine-applicable suggestions applied one at a time | finding 29: 111 lint fixes break builds (six shapes reduced); error-recovery suggestions that leave the error or do not parse noted | -| diagnostic invariants (13) | `diag-check.py` | 18,374 tests | finding 30: debug output in two diagnostics; spans all in bounds | -| determinism (15) | `repro-diff.py` | 6,886 tests × repeat, other directory with `--remap-path-prefix`, `-Zthreads=8`, decoy `-L` library | nothing new: only `-Zthreads` differences, all in the known async fn (#162202) and RPIT (#163878) families | -| feature gates (17) | `gate-check.py` | 143 unstable attributes × 14 positions; 156 unstable library items with resolvable paths × use, renamed use, glob, impl, value, type | every library spelling gated; finding 31 (an ICE after the gate error for `#[rustc_main]` on non-functions); `#[feature]` outside the crate root only warns (intended) | +| suggestions apply (18) | `mirth-lab suggest-diff` | 17,945 tests without `run-rustfix`, 7,385 machine-applicable suggestions applied one at a time | finding 29: 111 lint fixes break builds (six shapes reduced); error-recovery suggestions that leave the error or do not parse noted | +| diagnostic invariants (13) | `mirth-lab diag-check` | 18,374 tests | finding 30: debug output in two diagnostics; spans all in bounds | +| determinism (15) | `mirth-lab repro-diff` | 6,886 tests × repeat, other directory with `--remap-path-prefix`, `-Zthreads=8`, decoy `-L` library | nothing new: only `-Zthreads` differences, all in the known async fn (#162202) and RPIT (#163878) families | +| feature gates (17) | `mirth-lab gate-check` | 143 unstable attributes × 14 positions; 156 unstable library items with resolvable paths × use, renamed use, glob, impl, value, type | every library spelling gated; finding 31 (an ICE after the gate error for `#[rustc_main]` on non-functions); `#[feature]` outside the crate root only warns (intended) | + +## Running the checks + +The checks are subcommands of `mirth-lab` (`crates/mirth-lab`; `mirth-lab --help` lists them): + +```sh +cargo build --release -p mirth-lab +R=~/mirth-work/campaign/rustc/bin/rustc T=~/mirth-work/rust/tests/ui +target/release/mirth-lab opt-diff --rustc $R --cranelift "$(rustup +nightly-2026-10-06 which rustc)" --tests $T --work +target/release/mirth-lab solver-diff --rustc $R --tests $T --work +target/release/mirth-lab abi-diff --rustc $R --rust ~/mirth-work/rust --work --seed 3 +target/release/mirth-lab release-diff --corpus ~/proofhouse-repos/rust --old nightly-2026-07-18 --new nightly-2026-10-06 --work +``` + +Sweeps over UI tests share `--tests`, `--work`, `--only`, `--known`, `--jobs`, `--recheck` and +`--pause-on-finding` (exit 3 at the first finding: the frontier loop). Results go to +`/results.jsonl`, findings to `/findings//`. diff --git a/docs/hunt.md b/docs/hunt.md index 2f0d1c3..4a02cd5 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -44,7 +44,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 18 | after a fatal error (a missing lang item), an incremental rebuild reports fewer errors than a clean build: the fatal error is reached in a different query order | diagnostics only; found by the UI-test fuzzer; stock nightly; [facts](hunt/fatal-error-order.md); labelled known in `ui-fuzz.py` | | 19 | on riscv64 and loongarch64, an `extern "C"` call passes an `i32` (or narrower integer) that lands on the stack without sign-extending it; a clang-compiled callee reads the slot as already extended | **looks new**; ABI, stable code; found by the ABI differential against clang ([`checks.md`](checks.md)); since at least 1.80; cause found (extension only `if *avail_gprs >= 1` in `callconv/riscv.rs`, same in `loongarch.rs`); [facts](hunt/riscv-stack-arg-extension.md) | | 20 | on RISC-V and LoongArch hard-float targets, a `repr(C)` struct of one float and one pointer is passed in a floating-point and an integer register; clang passes it by the integer convention, so C and Rust disagree on where it is | **looks new**; ABI, stable code; found by the ABI differential; since at least 1.80; cause found (`Primitive::Pointer` counted as an integer in `should_use_fp_conv_helper`, `callconv/riscv.rs` and `loongarch.rs`); [facts](hunt/riscv-float-pointer-struct.md) | -| 21 | `-Zvalidate-mir` rejects MIR the compiler builds from accepted code: projections into `#[repr(simd)]` types (banned by MCP#838) in 9 SIMD tests, and an unsize coercion to `Pin>` in `async-await/issue-86507.rs` | found by the internal-checks sweep (`crash-diff.py`); stock nightly with `-Zvalidate-mir`; compiletest does not validate UI tests; [facts](hunt/internal-checks.md) | +| 21 | `-Zvalidate-mir` rejects MIR the compiler builds from accepted code: projections into `#[repr(simd)]` types (banned by MCP#838) in 9 SIMD tests, and an unsize coercion to `Pin>` in `async-await/issue-86507.rs` | found by the internal-checks sweep (`mirth-lab crash-diff`); stock nightly with `-Zvalidate-mir`; compiletest does not validate UI tests; [facts](hunt/internal-checks.md) | | 22 | the new trait solver trips a debug assertion in region outlives (`regions.rs:37`, `!type_outlives.has_non_rigid_aliases()`) on 5 UI tests | debug-assertion builds with nightly's default solver; hidden in CI by compiletest's solver pin; a sibling of closed #160206; [facts](hunt/internal-checks.md) | | 23 | an `attempt to add with overflow` in `ty/instance.rs:421` compiling `recursion/issue-83150.rs` under the new solver | overflow-checked builds; hidden by the solver pin; [facts](hunt/internal-checks.md) | | 24 | `-Zvalidate-mir` rejects a move of a dereferenced unsized place into a call (`unsized-locals/unsized-exprs2.rs`) | incomplete `unsized_fn_params`; [facts](hunt/internal-checks.md) | diff --git a/docs/hunt/glob-ambiguity-order.md b/docs/hunt/glob-ambiguity-order.md index 8f2d388..2f1d10d 100644 --- a/docs/hunt/glob-ambiguity-order.md +++ b/docs/hunt/glob-ambiguity-order.md @@ -1,6 +1,6 @@ # Glob-import ambiguity depends on the order of items -Facts for finding 28. Found by the equivalent-rewrite differential (`rustc/rewrite-diff.py`, the +Facts for finding 28. Found by the equivalent-rewrite differential (`mirth-lab rewrite-diff`, the `reorder` rewrite: top-level items in reverse order, `use` items first) on `tests/ui/imports/ambiguous-9.rs`; the same rewrite also flips `imports/ambiguous-14.rs` (error → compiles) and `imports/overwrite-different-ambig-2.rs` (compiles → error). diff --git a/docs/hunt/internal-checks.md b/docs/hunt/internal-checks.md index 1af986d..8e75653 100644 --- a/docs/hunt/internal-checks.md +++ b/docs/hunt/internal-checks.md @@ -1,6 +1,6 @@ # rustc's internal checks on the UI tests -Facts for findings 21 to 24. Found by `rustc/crash-diff.py` ([`checks.md`](../checks.md), check 19): +Facts for findings 21 to 24. Found by `mirth-lab crash-diff` ([`checks.md`](../checks.md), check 19): every standalone UI test (18,624) compiled with the release compiler under test and again with a compiler built from the same tree with `rust.debug-assertions = true`, `rust.debug-assertions-std = true` and `rust.overflow-checks = true`, plus `-Zvalidate-mir`. @@ -80,4 +80,4 @@ rust.debug-assertions-std=true --set rust.overflow-checks=true` (mirth: `~/mirth ## Local stopgap None: these are checks failing, not wrong output, and they do not affect mirth's incremental -checks. `crash-diff.py` takes them as `--known` (`rustc/crash-known.txt`). +checks. `mirth-lab crash-diff` takes them as `--known` (`rustc/crash-known.txt`). diff --git a/docs/hunt/lint-fixes-break-builds.md b/docs/hunt/lint-fixes-break-builds.md index 17230aa..fd1f67b 100644 --- a/docs/hunt/lint-fixes-break-builds.md +++ b/docs/hunt/lint-fixes-break-builds.md @@ -1,6 +1,6 @@ # Machine-applicable lint fixes that break builds -Facts for finding 29. Found by the suggestions-apply check (`rustc/suggest-diff.py`, check 18 in +Facts for finding 29. Found by the suggestions-apply check (`mirth-lab suggest-diff`, check 18 in [`checks.md`](../checks.md)): every `MachineApplicable` suggestion of every UI test without `//@ run-rustfix` (17,945 tests, 7,385 suggestions), each applied alone and compiled again. A lint's machine-applicable fix is what `cargo fix` and `cargo clippy --fix` apply without asking; diff --git a/docs/hunt/promoted-validation-generic.md b/docs/hunt/promoted-validation-generic.md index 9a4dbaf..10aea4f 100644 --- a/docs/hunt/promoted-validation-generic.md +++ b/docs/hunt/promoted-validation-generic.md @@ -1,6 +1,6 @@ # An invalid constant is accepted when its unused reference sits in a generic function -Facts for finding 25. Found by the equivalent-rewrite differential (`rustc/rewrite-diff.py`, the +Facts for finding 25. Found by the equivalent-rewrite differential (`mirth-lab rewrite-diff`, the `generic-wrap` rewrite: a function's body moved into a generic inner function called with `()`) on `tests/ui/consts/interior-mut-const-via-union.rs`. @@ -68,4 +68,4 @@ parameter it gives E0080 at every level. ## Local stopgap -None; not an incremental difference. `rewrite-diff.py` lists the test as known for `generic-wrap`. +None; not an incremental difference. `mirth-lab rewrite-diff` lists the test as known for `generic-wrap`. diff --git a/docs/hunt/release-regressions.md b/docs/hunt/release-regressions.md index b5a44cc..2382e22 100644 --- a/docs/hunt/release-regressions.md +++ b/docs/hunt/release-regressions.md @@ -1,6 +1,6 @@ # Regressions in real crates, nightly-2026-07-18 to nightly-2026-10-06 -Facts for findings 26 and 27. Found by the release-to-release check (`rustc/release-diff.py`, +Facts for findings 26 and 27. Found by the release-to-release check (`mirth-lab release-diff`, [`checks.md`](../checks.md) check 2): `cargo check --locked` of 87 popular repositories (`~/proofhouse-repos/rust`) under both nightlies. 53 behave the same, 18 fail on both, 10 could not fetch their locked dependencies. Of the 5 regressions, two are `allocative 0.3.4`, which enables diff --git a/docs/hunt/riscv-float-pointer-struct.md b/docs/hunt/riscv-float-pointer-struct.md index ff76089..db475ec 100644 --- a/docs/hunt/riscv-float-pointer-struct.md +++ b/docs/hunt/riscv-float-pointer-struct.md @@ -1,7 +1,7 @@ # RISC-V and LoongArch: a struct of a float and a pointer goes in the wrong registers Facts for finding 20. Found by the ABI differential ([`checks.md`](../checks.md), check 14: -`rustc/abi-diff.py`). +`mirth-lab abi-diff`). ## What happens diff --git a/docs/hunt/riscv-stack-arg-extension.md b/docs/hunt/riscv-stack-arg-extension.md index 5bf73ad..303166b 100644 --- a/docs/hunt/riscv-stack-arg-extension.md +++ b/docs/hunt/riscv-stack-arg-extension.md @@ -1,7 +1,7 @@ # riscv64 and loongarch64: integer arguments passed on the stack are not sign-extended Facts for finding 19. Found by the ABI differential ([`checks.md`](../checks.md), check 14: -`rustc/abi-diff.py`, rustc's `extern "C"` lowering against clang's for random C signatures). +`mirth-lab abi-diff`, rustc's `extern "C"` lowering against clang's for random C signatures). ## What happens diff --git a/docs/solver-triage.md b/docs/solver-triage.md index 9334218..95fc352 100644 --- a/docs/solver-triage.md +++ b/docs/solver-triage.md @@ -6,7 +6,7 @@ solver everywhere, since #160895), checked against the tracking issue affected-crates tables and the "unintended breakage" list), rust-lang/rust issues and rust-lang/trait-system-refactor-initiative (tsri) issues, on 2026-10-10. -Sources: the UI-test solver differential ([`solver.md`](solver.md), `rustc/solver-diff.py`: +Sources: the UI-test solver differential ([`solver.md`](solver.md), `mirth-lab solver-diff`: 26 tests accepted by the old solver and rejected by the new, 3 crashes), the internal-checks sweep (findings 22 and 23), release-to-release (finding 27). @@ -30,7 +30,7 @@ Sources: the UI-test solver differential ([`solver.md`](solver.md), `rustc/solve | F | higher-ranked associated type no longer guides inference (`escaping-bounds`; `diskann-wide`) | 1 + crate | yes | #160895, tsri#168: intended | low | surrealdb as an affected project on #160895 | | G | type alias `impl Trait`: "does not constrain", a cycle | 2 | no | #160895: RPIT/TAIT handling changed | low | no | | H | `fn_delegation` with `impl Trait` returns: E0282 | 1 | no (incomplete) | no | low | optional | -| I | a help suggestion containing inference variables: "consider casting both fn items to fn pointers using `as fn(?0t) -> ?0t`" (`fn/fn_def_opaque_coercion_to_fn_ptr.rs`; the old solver gives no such help) | 1 | yes | no | low | optional (found by `diag-check.py`) | +| I | a help suggestion containing inference variables: "consider casting both fn items to fn pointers using `as fn(?0t) -> ?0t`" (`fn/fn_def_opaque_coercion_to_fn_ptr.rs`; the old solver gives no such help) | 1 | yes | no | low | optional (found by `mirth-lab diag-check`) | | 22 | debug assertion `!type_outlives.has_non_rigid_aliases()` in region outlives | 5 | some | sibling of closed #160206 | low | optional: debug builds only, but an invariant broken | | 23 | integer overflow in `ty/instance.rs:421` (`recursion/issue-83150.rs`) | 1 | yes | no | low | optional | diff --git a/docs/solver.md b/docs/solver.md index b28ca82..7bfb49b 100644 --- a/docs/solver.md +++ b/docs/solver.md @@ -5,11 +5,11 @@ everywhere by default (`CFG_DEFAULT_NEXT_SOLVER_GLOBALLY`), while compiletest pa `-Znext-solver=coherence` to every UI test, so the suite checks the old solver only. Code that nightly users compile is therefore not what the suite checks. -`rustc/ui-solver-diff.py` compiles every UI test both ways with stock nightly-2026-10-06, the way +`rustc/ui-mirth-lab solver-diff` compiles every UI test both ways with stock nightly-2026-10-06, the way its `//@` headers say (tests needing auxiliary crates or another target, and tests that set `-Znext-solver` themselves, are left out): - rustc/ui-solver-diff.py --rustc --tests /tests/ui --out + rustc/ui-mirth-lab solver-diff --rustc --tests /tests/ui --out 17,716 tests compiled both ways; **300 differ**: 271 fail either way with a different first error, **26 compile under the pinned solver and fail under nightly's default**, and **3 crash under the diff --git a/rustc/abi-diff.py b/rustc/abi-diff.py deleted file mode 100644 index b49508d..0000000 --- a/rustc/abi-diff.py +++ /dev/null @@ -1,468 +0,0 @@ -#!/usr/bin/env python3 -"""ABI differential: rustc's `extern "C"` must lower a signature the way clang lowers the same C -signature, on every target both support. - -Generates random C signatures (bool, integers of each width, float, double, pointers, __int128 -on 64-bit targets, and repr(C) structs, unions and arrays of them, nested, packed or -over-aligned), writes each as a C function (compiled by clang for the target's LLVM triple) and -as a Rust `#[no_mangle] extern "C" fn` (compiled by rustc for the target against minicore, so no -sysroot is needed), and compares the two LLVM IR signatures, parameter by parameter after -first-class aggregates are flattened (LLVM assigns their elements to registers one by one): - - finding a parameter or the return value differs in its register class (integer, floating - point, vector, memory), in an extension attribute (zeroext, signext), in inreg, - byval or sret, in the alignment of a byval or sret pointer, in the number of - parameters, or in the calling convention - note the same register classes with different IR types (`{double, double}` against - `float, double`), or a `noundef` difference - - rustc/abi-diff.py --rustc --rust --work [--targets t1,t2 | --all] - [--count 200] [--seed 1] [--jobs 8] [--clang clang] - -Known bugs are labelled, not reported: rust-lang/rust#163911 (x86_64 bool returns) and findings -19 and 20 in docs/hunt.md (RISC-V and LoongArch); so are two differences this host cannot decide -(i686 MSVC small-struct returns, a PowerPC64 `inreg` float). - -Writes //{a.c,a.rs,c.ll,r.ll} and /results.json (per target: functions -compared, findings, notes), and prints the findings grouped by kind. -""" - -import argparse -import ast -import json -import random -import re -import subprocess -from collections import defaultdict -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path -import os - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--rust", required=True) -p.add_argument("--work", required=True) -p.add_argument("--targets") -p.add_argument("--all", action="store_true", help="every target rustc knows; the default is MAIN, the tier 1 and " - "2 targets whose differences have been triaged (the others still show representation differences)") -p.add_argument("--count", type=int, default=200) -p.add_argument("--seed", type=int, default=1) -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--clang", default="clang") -args = p.parse_args() -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) -ENV = dict(os.environ, RUSTC_BOOTSTRAP="1") - -SCALARS = [ # (C, Rust) - ("_Bool", "bool"), ("signed char", "i8"), ("unsigned char", "u8"), ("short", "i16"), - ("unsigned short", "u16"), ("int", "i32"), ("unsigned int", "u32"), ("long long", "i64"), - ("unsigned long long", "u64"), ("float", "f32"), ("double", "f64"), ("void*", "*mut u8"), -] -WIDE = [("__int128", "i128"), ("unsigned __int128", "u128")] - - -class Gen: - def __init__(self, rng, wide): - self.rng, self.wide, self.structs, self.names = rng, wide, [], 0 - - def scalar(self): - pool = SCALARS + (WIDE if self.wide else []) - return self.rng.choice(pool) - - def field_type(self, depth): - r = self.rng.random() - if depth < 2 and r < 0.15: - return self.aggregate(depth + 1) - if r < 0.25: - c, rs = self.scalar() - n = self.rng.choice([1, 2, 3, 4, 8]) - return ("array", c, rs, n) - return self.scalar() - - def aggregate(self, depth=0): - name = f"S{self.names}" - self.names += 1 - union = self.rng.random() < 0.12 - packed = not union and self.rng.random() < 0.08 - # Rust rejects packed with align, and a packed type containing an over-aligned one: - # packed structs get scalars and arrays only. - align = None if packed else self.rng.choice([None] * 9 + [16, 32]) - fields = [self.field_type(2 if packed else depth) for _ in range(self.rng.randint(1, 5))] - self.structs.append((name, union, packed, align, fields)) - return (name, name) - - def ty(self): - return self.aggregate() if self.rng.random() < 0.4 else self.scalar() - - -def c_field(i, f): - if f[0] == "array": - return f"{f[1]} f{i}[{f[3]}];" - c = f[0] - return f"{c} f{i};" - - -def r_field(i, f): - if f[0] == "array": - return f"pub f{i}: [{f[2]}; {f[3]}]," - return f"pub f{i}: {f[1]}," - - -def c_ty(t): - return t[0] - - -def program(target_wide, seed): - rng = random.Random(seed) - g = Gen(rng, target_wide) - fns = [] - for k in range(args.count): - params = [g.ty() for _ in range(rng.randint(0, 8))] - ret = None if rng.random() < 0.15 else g.ty() - fns.append((f"f{k}", params, ret)) - c = ["#include "] - rs = ["#![feature(no_core)]", "#![no_core]", '#![crate_type = "lib"]', - "#![allow(improper_ctypes_definitions, unused, non_snake_case)]", - "extern crate minicore;", "use minicore::*;"] - for name, union, packed, align, fields in g.structs: - kw = "union" if union else "struct" - attrs = (" __attribute__((packed))" if packed else "") + (f" __attribute__((aligned({align})))" if align else "") - c.append(f"typedef {kw} {name} {{ {' '.join(c_field(i, f) for i, f in enumerate(fields))} }}{attrs} {name};") - repr_ = "C" + (", packed" if packed else "") + (f", align({align})" if align else "") - rs.append(f"#[repr({repr_})] pub {kw} {name} {{ {' '.join(r_field(i, f) for i, f in enumerate(fields))} }}") - # Union fields must be Copy; minicore has no derive. - rs.append(f"impl Copy for {name} {{}}") - for name, params, ret in fns: - cp = ", ".join(f"{c_ty(t)} a{i}" for i, t in enumerate(params)) or "void" - c.append(f"{c_ty(ret) if ret else 'void'} {name}({cp}) {{ for (;;); }}") - rp = ", ".join(f"a{i}: {t[1]}" for i, t in enumerate(params)) - rr = f" -> {ret[1]}" if ret else "" - rs.append(f"#[no_mangle] pub extern \"C\" fn {name}({rp}){rr} {{ loop {{}} }}") - return "\n".join(c) + "\n", "\n".join(rs) + "\n", fns - - -DEFINE = re.compile(r"^define\s+(.*?)@(\w+)\((.*)\)(.*)\{\s*$") -DROP = re.compile(r"\b(noundef|nonnull|noalias|nocapture|readonly|readnone|writeonly|writable|dead_on_unwind|" - r"captures\([^)]*\)|dereferenceable(_or_null)?\(\d+\)|initializes\([^)]*\)|range\([^)]*\)|" - r"nofpclass\([^)]*\)|immarg|returned|local_unnamed_addr|unnamed_addr|dso_local|" - r"dso_preemptable|hidden|protected|internal|private|nounwind|noinline|optnone|" - r"!\w+ !\d+|#\d+)\b") - - -def split_top(s): - out, depth, cur = [], 0, "" - for ch in s: - if ch in "({[<": - depth += 1 - elif ch in ")}]>": - depth -= 1 - if ch == "," and depth == 0: - out.append(cur.strip()) - cur = "" - else: - cur += ch - if cur.strip(): - out.append(cur.strip()) - return out - - -def param_type(p): - """The IR type of a parameter declaration, its attributes, and whether it has noundef.""" - p = re.sub(r"\s+%[\w.]+$", "", p.strip()) - noundef = "noundef" in p - m = re.match(r"^(\{[^}]*\}|\[[^\]]*\]|<[^>]*>|[\w.%*]+)(.*)$", p) - ty, attrs = (m.group(1), m.group(2)) if m else (p, "") - attrs = DROP.sub("", attrs) - kept = [] - for a in re.findall(r"(zeroext|signext|inreg|byval\([^)]*\)|sret\([^)]*\)|byref\([^)]*\)|align \d+|inalloca\([^)]*\))", attrs): - kept.append(re.sub(r"\(.*\)$", "", a) if a.startswith(("byval", "sret", "byref", "inalloca")) else a) - # An `align` on a plain pointer is a hint, not ABI; on byval and sret it is ABI. - if not any(k.startswith(("byval", "sret", "byref")) for k in kept): - kept = [k for k in kept if not k.startswith("align")] - return ty, tuple(sorted(kept)), noundef - - -def flatten(ty, named): - """A type's register-assignable parts; `named` maps the module's %struct names to bodies.""" - ty = ty.strip() - if ty in named: - ty = named[ty] - if ty.startswith("<{") and ty.endswith("}>"): # packed struct - ty = ty[1:-1] - if ty.startswith("{") and ty.endswith("}"): - return [x for part in split_top(ty[1:-1]) for x in flatten(part.strip(), named)] - m = re.fullmatch(r"\[1 x (.*)\]", ty) - if m: # a one-element array goes where its element goes - return flatten(m.group(1), named) - return [ty] - - -def units(types, arch=None, ret=False, width=64): - """Register classes, one per register-sized unit: an integer wider than a register takes - several. For a return value, an array is its elements (returned in consecutive registers).""" - out = [] - for t in types: - m = re.fullmatch(r"\[(\d+) x (i\d+|float|double)\]", t) - mi = re.fullmatch(r"i(\d+)", t) - if mi and int(mi.group(1)) > width: - out += ["int"] * (int(mi.group(1)) // width) - # A float array is a homogeneous aggregate: in a return, or an ARM VFP argument, it - # takes consecutive floating-point registers like a struct of its elements. - elif m and (ret or (arch == "arm" and m.group(2) in ("float", "double"))): - out += units([m.group(2)] * int(m.group(1)), arch, ret, width) - elif t == "agg": - out.append("agg") - else: - k = klass(t) - # x86's SSE registers hold floats, doubles and vectors alike. - if arch in ("x86_64", "x86") and k in ("fp", "vec"): - k = "sse" - out.append(k) - return out - - -def klass(ty): - if ty in ("float", "double", "half", "bfloat", "fp128", "x86_fp80", "ppc_fp128") or ty.startswith("<"): - return "fp" if not ty.startswith("<") else "vec" - if ty.startswith("["): - m = re.match(r"\[(\d+) x (.*)\]", ty) - return f"[{m.group(1)} x {klass(m.group(2))}]" if m else ty - if ty == "void": - return "void" - return "int" - - -def signature(line, named): - m = DEFINE.match(line) - if not m: - return None - head, name, params, tail = m.groups() - cc = re.findall(r"\b(\w+cc|cc \d+)\b", head) - head = DROP.sub("", head).strip() - tm = re.search(r"(\{[^{}]*\}|<\{[^{}]*\}>|\[[^\]]*\]|<[^>]*>|[\w.%*]+)\s*$", head) - ret_ty = tm.group(1) if tm else "void" - ret_attrs = tuple(sorted(a for a in head[:tm.start() if tm else 0].split() if a in ("zeroext", "signext", "inreg"))) - ps = [] - for p in split_top(params): - ty, attrs, noundef = param_type(p) - for t in flatten(ty, named): - ps.append((t, attrs, noundef)) - rets = flatten(ret_ty, named) - return {"name": name, "cc": tuple(cc), "ret": rets, "ret_attrs": ret_attrs, "params": ps, - "ret_noundef": "noundef" in m.group(1)} - - -def signatures(ll): - named = {m.group(1): m.group(2).strip() for m in re.finditer(r"^(%[\w.]+) = type (.*)$", ll, re.M)} - return {s["name"]: s for s in (signature(l, named) for l in ll.splitlines() if l.startswith("define")) if s} - - -def compare(c, r, arch, slot): - """`arch`: the target's arch; `slot`: its stack slot size in bytes (the pointer width).""" - findings, notes = [], [] - if c["cc"] != r["cc"]: - findings.append(f"calling convention: clang {c['cc']} rustc {r['cc']}") - if c["ret_attrs"] != r["ret_attrs"]: - what = f"return attributes: clang {c['ret_attrs']} rustc {r['ret_attrs']}" - # x86's psABI leaves the bits above a small integer return undefined (rustc stopped - # extending them in #142389); only bool's bits 1-7 must be zero (#163911). - if arch in ("x86_64", "x86") and c["ret"] != ["i1"]: - notes.append(what + " (x86 psABI: upper bits undefined)") - else: - findings.append(what) - if units(c["ret"], arch, True, slot * 8) != units(r["ret"], arch, True, slot * 8): - findings.append(f"return: clang {c['ret']} rustc {r['ret']}") - elif c["ret"] != r["ret"]: - notes.append(f"return types: clang {c['ret']} rustc {r['ret']}") - cp, rp = c["params"], r["params"] - # ARM and 64-bit PowerPC split a byval aggregate between registers and the stack as they do - # an array argument of the same size: both are "an aggregate". - if arch in ("arm", "powerpc64"): - agg = lambda t, a: ("agg", (), False) if ("byval" in a or re.match(r"\[\d+ x i\d+\]", t)) else (t, a, None) - cp = [agg(t, a) if agg(t, a)[0] == "agg" else (t, a, n) for t, a, n in cp] - rp = [agg(t, a) if agg(t, a)[0] == "agg" else (t, a, n) for t, a, n in rp] - if units([t for t, _, _ in cp], arch, False, slot * 8) != units([t for t, _, _ in rp], arch, False, slot * 8): - what = f"parameters: clang {[t for t, _, _ in cp]} rustc {[t for t, _, _ in rp]}" - # i386 passes every argument on the stack: a struct expanded into its scalars and a - # byval copy of it are the same bytes. - if arch == "x86" and any("byval" in a for _, a, _ in cp + rp): - notes.append(what + " (x86: same stack bytes)") - else: - findings.append(what) - elif [t for t, _, _ in cp] != [t for t, _, _ in rp]: - notes.append(f"parameter types: clang {[t for t, _, _ in cp]} rustc {[t for t, _, _ in rp]}") - elif True: - for i, ((ct, ca, cn), (rt, ra, rn)) in enumerate(zip(cp, rp)): - if ca != ra: - what = f"parameter {i} attributes: clang {ca} rustc {ra} ({ct})" - aligns = [int(a.split()[1]) for a in ca + ra if a.startswith("align ")] - rest_c = [a for a in ca if not a.startswith("align ")] - rest_r = [a for a in ra if not a.startswith("align ")] - ext_only_rust = not rest_c and set(rest_r) <= {"zeroext", "signext"} - if rest_c == rest_r and aligns and max(aligns) <= slot: - # A byval copy goes in stack slots at least `slot` bytes aligned either way. - notes.append(what + " (both within a stack slot)") - elif arch in ("wasm32", "wasm64") and {*rest_c, *rest_r} <= {"byval"}: - # WebAssembly lowers byval to a pointer to a copy the caller makes. - notes.append(what + " (wasm: byval is a pointer to a copy)") - elif arch == "x86" and {*rest_c, *rest_r} <= {"byval"}: - # i386 passes everything on the stack: a byval copy and a struct expanded - # into its scalars occupy the same bytes. - notes.append(what + " (x86: same stack bytes)") - elif arch == "x86_64" and ext_only_rust: - # Win64: rustc extends small integers, clang does not; LLVM does not rely on - # it in the callee (checked: both re-extend), so nothing observable. - notes.append(what + " (win64: extension not relied on)") - else: - findings.append(what) - if ct != rt: - notes.append(f"parameter {i}: clang {ct} rustc {rt}") - if cn != rn: - notes.append(f"parameter {i} noundef: clang {cn} rustc {rn}") - return findings, notes - - -MAIN = ("x86_64-unknown-linux-gnu x86_64-unknown-linux-musl x86_64-pc-windows-msvc x86_64-pc-windows-gnu " - "x86_64-apple-darwin i686-unknown-linux-gnu i686-pc-windows-msvc aarch64-unknown-linux-gnu " - "aarch64-apple-darwin aarch64-pc-windows-msvc aarch64-unknown-linux-musl armv7-unknown-linux-gnueabihf " - "arm-unknown-linux-gnueabi thumbv7em-none-eabihf riscv64gc-unknown-linux-gnu riscv32imac-unknown-none-elf " - "loongarch64-unknown-linux-gnu powerpc64le-unknown-linux-gnu s390x-unknown-linux-gnu wasm32-unknown-unknown " - "wasm32-wasip1").split() - - -def known(arch, finding): - """A finding that is a bug already recorded: the label, or None.""" - if arch == "x86_64" and finding.startswith("return attributes: clang ('zeroext',)"): - return "rust-lang/rust#163911" - if arch in ("riscv64", "riscv32", "loongarch64"): - if re.match(r"parameter \d+ attributes: clang \('(signext|zeroext)',\) rustc \(\)", finding): - return "finding 19 (docs/hunt.md)" - if finding.startswith(("parameters:", "return:")): - c, _, r = finding.partition(" rustc ") - fp = lambda text: len(re.findall(r"'(float|double)'", text)) - if fp(r) > fp(c): - return "finding 20 (docs/hunt.md)" - return None - - -def unresolved(arch, target, finding): - """A difference whose correct side needs a reference this host lacks: the label, or None.""" - lists = re.fullmatch(r"parameters: clang (\[.*?\]) rustc (\[.*\])", finding) - sret_only = bool(lists) and (lambda c, r: c[:1] == ["ptr"] and c[1:] == r)(*map(ast.literal_eval, lists.groups())) - if target.endswith("windows-msvc") and arch == "x86" and ( - (finding.startswith("return:") and "'void'" in finding) or sret_only): - # clang returns an 8-byte struct with an array field of 3 bytes indirectly (its - # register-size rule recurses into fields); rustc and MSVC's documentation return any - # 8-byte struct in edx:eax. Needs MSVC to decide. - return "i686 msvc small-struct return (needs MSVC)" - if arch == "powerpc64" and re.match(r"parameter \d+ attributes: clang \('inreg',\) rustc \(\) \((float|double)\)", finding): - # clang marks a float from a single-member aggregate inreg; whether the PowerPC backend - # then places it differently needs a run on the target. - return "ppc64 inreg float (needs a run)" - return None - - -def run(argv, cwd): - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=600, cwd=cwd, env=ENV) - return r.returncode, r.stderr - except subprocess.TimeoutExpired: - return -1, "timeout" - - -def spec(target): - out = subprocess.run([args.rustc, "--print", "target-spec-json", "-Zunstable-options", "--target", target], - capture_output=True, text=True, env=ENV) - return json.loads(out.stdout) if out.returncode == 0 else None - - -def one_target(target): - s = spec(target) - if not s: - return target, {"skip": "no target spec"} - width = int(s.get("target-pointer-width", 64)) - if width < 32 or s.get("c-int-width", "32") != "32": - return target, {"skip": "16-bit int"} - d = WORK / target - d.mkdir(parents=True, exist_ok=True) - csrc, rsrc, fns = program(width == 64 and s.get("arch") not in ("sparc64",), args.seed) - (d / "a.c").write_text(csrc) - (d / "a.rs").write_text(rsrc) - base = [args.rustc, "--target", target, "-Zunstable-options", "--edition", "2021", "-Cpanic=abort", - "--out-dir", str(d)] - code, err = run(base + ["--crate-type", "rlib", "--crate-name", "minicore", "-Awarnings", - str(Path(args.rust) / "tests/auxiliary/minicore.rs")], d) - if code: - return target, {"skip": "minicore does not build", "err": err[-500:]} - code, err = run(base + ["--emit=llvm-ir", "-Copt-level=0", "--extern", f"minicore={d}/libminicore.rlib", - "-o", str(d / "r.ll"), str(d / "a.rs")], d) - if code: - return target, {"skip": "rust side does not build", "err": err[-1500:]} - cflags = [] - # The target's CPU and features decide parts of the ABI in clang too (soft-float, SSE). - if s.get("cpu") and s["cpu"] != "generic": - cflags += ["-Xclang", "-target-cpu", "-Xclang", s["cpu"]] - for feature in filter(None, s.get("features", "").split(",")): - cflags += ["-Xclang", "-target-feature", "-Xclang", feature] - if s.get("llvm-abiname"): - cflags.append(f"-mabi={s['llvm-abiname']}") - if s.get("llvm-floatabi") == "hard": - cflags.append("-mfloat-abi=hard") - code, err = run([args.clang, f"--target={s['llvm-target']}", "-ffreestanding", "-S", "-emit-llvm", "-O0", - "-Wno-everything", *cflags, "-o", str(d / "c.ll"), str(d / "a.c")], d) - if code: - return target, {"skip": "clang does not build", "err": err[-800:]} - cs, rs = signatures((d / "c.ll").read_text()), signatures((d / "r.ll").read_text()) - findings, notes, known_hits = {}, {}, {} - for name, _, _ in fns: - if name not in cs or name not in rs: - continue - f, n = compare(cs[name], rs[name], s.get("arch"), width // 8) - labelled = [(x, known(s.get("arch"), x) or unresolved(s.get("arch"), target, x)) for x in f] - f = [x for x, k in labelled if not k] - for x, k in labelled: - if k: - known_hits.setdefault(k, 0) - known_hits[k] += 1 - if f: - findings[name] = f - if n: - notes[name] = n - return target, {"compared": len(fns), "findings": findings, "notes": notes, "known": known_hits} - - -def main(): - if args.targets: - targets = args.targets.split(",") - elif not args.all: - targets = MAIN - else: - targets = subprocess.run([args.rustc, "--print", "target-list"], capture_output=True, text=True, - env=ENV).stdout.split() - with ThreadPoolExecutor(args.jobs) as ex: - results = dict(ex.map(one_target, targets)) - (WORK / "results.json").write_text(json.dumps(results, indent=1)) - kinds = defaultdict(lambda: defaultdict(int)) - skipped = defaultdict(list) - for t, r in results.items(): - if "skip" in r: - skipped[r["skip"]].append(t) - continue - for fs in r["findings"].values(): - for f in fs: - kinds[re.sub(r"\(.*|:.*", "", f).strip()][t] += 1 - compared = [t for t, r in results.items() if "skip" not in r] - knowns = defaultdict(int) - for r in results.values(): - for k, n in r.get("known", {}).items(): - knowns[k] += n - if knowns: - print("known: " + ", ".join(f"{k} {n}" for k, n in sorted(knowns.items()))) - print(f"{len(compared)} targets compared; skipped: " + ", ".join(f"{k} {len(v)}" for k, v in skipped.items())) - for kind, per in sorted(kinds.items(), key=lambda kv: -sum(kv[1].values())): - print(f"{kind}: {sum(per.values())} in {len(per)} targets: " - + ", ".join(f"{t} {n}" for t, n in sorted(per.items(), key=lambda kv: -kv[1])[:8])) - - -main() diff --git a/rustc/crash-diff.py b/rustc/crash-diff.py deleted file mode 100644 index c11f049..0000000 --- a/rustc/crash-diff.py +++ /dev/null @@ -1,93 +0,0 @@ -#!/usr/bin/env python3 -"""Internal checks on: what the compiler's own invariants say about every UI test. - -Compiles each standalone UI test with the compiler under test and again with a second compiler -built from the same source with debug assertions (`rust.debug-assertions`), with -`-Zvalidate-mir` added. Findings are a crash, a failed assertion or a MIR validation error -under the second that the first does not have: rustc's invariants failing where release -builds go on silently (19 of the last 1,000 ICE reports needed such a build). - - rustc/crash-diff.py --rustc --checked - --tests /tests/ui --work [--extra "-Zvalidate-mir"] [--only ] - [--known ] [--jobs 8] [--pause-on-finding] [--recheck] -""" - -import argparse -import json -import re -import shutil -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--checked", required=True) -p.add_argument("--extra", default="-Zvalidate-mir") -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def message(stderr): - """The first line saying what went wrong inside the compiler.""" - for pattern in (r"panicked at [^\n]*\n[^\n]*", r"internal compiler error: [^\n]*", r"broken MIR[^\n]*"): - m = re.search(pattern, stderr) - if m: - return re.sub(r"/\S+/compiler/", "compiler/", m.group(0))[:400] - return "" - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - emit = "metadata" if kind in ("check-pass", "check-fail", None) else "link" - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - a, ea, _ = uitest.compile(args.rustc, path.resolve(), d / "release", flags, edition, timeout=300, emit=emit) - b, eb, _ = uitest.compile(args.checked, path.resolve(), d / "checked", flags, edition, args.extra.split(), - timeout=600, emit=emit) - record = {"test": rel, "release": a, "checked": b} - found = [] - if b == "ice" and a != "ice": - found.append({"what": "only with internal checks", "message": message(eb), "stderr": eb[-4000:]}) - elif b == "ice" and a == "ice" and message(ea) != message(eb): - record["note"] = "both crash, differently" - record["found"] = [f"{f['what']}: {f['message'][:160]}" for f in found] - if found: - out = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - shutil.copy(path, out / path.name) - (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, - "extra": args.extra, "found": found}, indent=1)) - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, KINDS): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/diag-check.py b/rustc/diag-check.py deleted file mode 100644 index de83feb..0000000 --- a/rustc/diag-check.py +++ /dev/null @@ -1,171 +0,0 @@ -#!/usr/bin/env python3 -"""Diagnostic invariants: what every diagnostic rustc prints must satisfy, whatever the program. - -Compiles each standalone UI test with `--error-format=json` and checks every diagnostic -(children and suggestions included): - - internal user-facing text (message, labels, suggested code) contains compiler-internal - debug output: `DefId(`, region and type-variable debug names (`ReLateParam`, - `ReBound`, `ReVar`, `'{erased}`, `?0t`, `^0`), `Opaque(DefId`, `{closure#0}` in - suggested code - span a span outside its file: byte offsets past the end, lines past the last line, - a start after its end - nowhere an error without any span, other than summaries ("aborting due to") - duplicate the same diagnostic (level, code, message, primary span) twice (noted, not a - finding: some are blessed in .stderr files) - - rustc/diag-check.py --rustc --tests /tests/ui --work [--only ] - [--known ] [--jobs 8] [--pause-on-finding] [--recheck] -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -import sys -import tempfile -from collections import Counter -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) -INTERNAL = re.compile(r"DefId\(|\bRe(LateParam|Bound|Var|Early|Static)\b|'\{erased\}|\?\d+[tif]\b|" - r"'\^\d+(_\d+)?\b|Opaque\(DefId|\bAlias\((Projection|Opaque|Inherent|Free)|" - r"\bBoundRegionKind|\bDefPath\b|\bLocalDefId\b|\bTyKind::") -# In suggested code, compiler-made names that are not Rust. -INTERNAL_CODE = re.compile(r"\{closure#\d+\}|\{opaque#\d+\}|\{async block@|\{impl#\d+\}|\{constant#\d+\}") - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def walk(diag): - yield diag - for c in diag.get("children", []): - yield from walk(c) - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel} - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - argv = [args.rustc, str(path.resolve()), "--edition", edition or "2015", "--emit=metadata", "-o", - str(d / "x"), "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", - "--error-format=json", *flags] - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=120, cwd=d, - env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) - except subprocess.TimeoutExpired: - record["skip"] = "timeout" - return record, [] - if uitest.is_ice(r.stderr): - record["skip"] = "ice" - return record, [] - diags = [] - for line in r.stderr.splitlines(): - try: - diags.append(json.loads(line)) - except json.JSONDecodeError: - pass - files = {} - - def file_info(name): - if name not in files: - try: - data = Path(name).read_bytes() if Path(name).is_absolute() else (path.parent / name).read_bytes() - files[name] = (len(data), data.count(b"\n") + 1) - except OSError: - files[name] = None - return files[name] - - found, notes = [], [] - seen = Counter() - for diag in diags: - prim = next(((s["file_name"], s["byte_start"], s["byte_end"]) for s in diag.get("spans", []) - if s.get("is_primary")), None) - seen[(diag.get("level"), (diag.get("code") or {}).get("code"), diag["message"], prim)] += 1 - if (diag.get("level") == "error" and not diag.get("spans") and not diag.get("children") - and not re.match(r"aborting due to|could not compile|\d+ (previous )?errors?", diag["message"]) - and "#![feature" not in diag["message"]): - notes.append({"what": "nowhere", "message": diag["message"][:200]}) - for node in walk(diag): - texts = [node["message"]] + [s.get("label") or "" for s in node.get("spans", [])] - for t in texts: - m = INTERNAL.search(t) - if m: - found.append({"what": "internal", "token": m.group(0), "text": t[:300], - "code": (diag.get("code") or {}).get("code")}) - parts = [] - for s in node.get("spans", []): - repl = s.get("suggested_replacement") - if repl is not None: - m = INTERNAL.search(repl) or INTERNAL_CODE.search(repl) - if m: - found.append({"what": "internal", "token": m.group(0), "text": f"suggests {repl[:200]!r}", - "code": (diag.get("code") or {}).get("code")}) - parts.append((s["file_name"], s["byte_start"], s["byte_end"])) - info = file_info(s["file_name"]) if not s["file_name"].startswith("<") else None - if info: - size, lines = info - if (s["byte_start"] > s["byte_end"] or s["byte_end"] > size or s["line_start"] > lines - or s["line_end"] > lines or s["line_start"] > s["line_end"]): - found.append({"what": "span", "span": {k: s[k] for k in ("file_name", "byte_start", - "byte_end", "line_start", "line_end")}, "size": size, "lines": lines, - "message": node["message"][:200]}) - - for k, n in seen.items(): - if n > 1 and k[0] in ("error", "warning"): - notes.append({"what": "duplicate", "message": k[2][:200], "times": n}) - # One entry per distinct problem. - uniq = {json.dumps(f, sort_keys=True): f for f in found} - found = list(uniq.values()) - record["found"] = [f"{f['what']}: {f.get('token') or ''} {(f.get('text') or f.get('message') or '')[:100]}" for f in found] - record["notes"] = [f"{n['what']}: {n['message'][:100]}" for n in notes] - if found: - out = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - shutil.copy(path, out / path.name) - (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, - "found": found, "notes": notes}, indent=1)) - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - # Tests that ask for compiler internals on purpose: verbose printing, dump attributes. - debug = lambda text, flags: (any(re.search(r"verbose|-Zdump|unpretty|print-", f) for f in flags) - or re.search(r"#!?\[rustc_(dump|effective_visibility|regions|variance|" - r"outlives|layout|abi|def_path|symbol_name|object_lifetime_default|" - r"dump_[a-z_]+|evaluate_where_clauses|then_this_would_need|" - r"if_this_changed|clean|partition)", text) - or "assumptions_on_binders" in text) - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, KINDS, debug): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/gate-check.py b/rustc/gate-check.py deleted file mode 100644 index 04cffe6..0000000 --- a/rustc/gate-check.py +++ /dev/null @@ -1,200 +0,0 @@ -#!/usr/bin/env python3 -"""Feature gates: nothing unstable may be usable from stable code, whatever the spelling. - -Two enumerations, each compiled without any `#![feature]` and with `RUSTC_BOOTSTRAP` unset -(as a stable user would): - - attributes every attribute in the "Unstable attributes" part of - compiler/rustc_feature/src/builtin_attrs.rs, placed on each kind of item and - position (a function with and without a body, a trait method declaration, a - foreign function, a parameter, a struct, a field, an impl, a module, a closure, a - statement, the crate) - library every top-level `pub` item of core, alloc and std marked `#[unstable(feature)]`, - reached by `use` (directly, renamed, through a glob), implemented (traits) and - taken as a value (functions) - -A program must report the gate (E0658, or "is experimental" / "unstable"). One that compiles is a -finding; one that fails without mentioning the gate is noted (an earlier error may have hidden -it). Library items whose path does not resolve even with the feature are skipped. - - rustc/gate-check.py --rustc --rust --work [--jobs 8] - [--only attributes|library] -""" - -import argparse -import json -import os -import re -import subprocess -import tempfile -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--rust", required=True) -p.add_argument("--work", required=True) -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--only") -args = p.parse_args() -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) -RUST = Path(args.rust) -ENV = {k: v for k, v in os.environ.items() if k != "RUSTC_BOOTSTRAP"} -GATE = re.compile(r"E0658|is experimental|is unstable|unstable feature|use of unstable|" - r"internal implementation detail|unstable library feature|requires a nightly|" - r"used internally by the standard library|may not be used|are considered unstable|" - r"is an internal|cannot be used on stable") - -# Arguments for attributes that need them; the rest are written bare. -ATTR_ARGS = { - "optimize": "(speed)", "patchable_function_entry": "(prefix_nops = 1, entry_nops = 1)", - "instrument_fn": ' = "on"', "cfi_encoding": ' = "u1x"', "register_tool": "(mytool)", - "register_attribute_tool": "(mytool)", "register_lint_tool": "(mytool)", "linkage": ' = "weak"', - "lang": ' = "mirth_nonexistent"', "rustc_on_unimplemented": '(message = "x")', - "rustc_diagnostic_item": ' = "mirth_x"', "test_runner": "(crate::r)", "pattern_complexity_limit": " = 10", - "rustc_legacy_const_generics": "(0)", "rustc_layout_scalar_valid_range_start": "(1)", - "rustc_abi": "(debug)", "rustc_macro_transparency": ' = "semitransparent"', "unstable": '(feature = "x", issue = "none")', - "stable": '(feature = "x", since = "1.0.0")', "rustc_const_unstable": '(feature = "x", issue = "none")', - "rustc_const_stable": '(feature = "x", since = "1.0.0")', "feature": "(mirth_nonexistent)", - "rustc_objc_class": ' = "X"', "rustc_objc_selector": ' = "x"', "rustc_confusables": '("x")', - "rustc_must_implement_one_of": "(a, b)", "rustc_deprecated_safe_2024": "", - "allow_internal_unstable": "(core_intrinsics)", "rustc_allow_const_fn_unstable": "(x)", - "rustc_default_body_unstable": '(feature = "x", issue = "none")', "unstable_removed": "", - "rustc_simd_monomorphize_lane_limit": ' = "8"', "rustc_scalable_vector": "(4)", -} -# Each position: (name, program with {A} where the attribute goes). -POSITIONS = [ - ("fn", "{A}\npub fn f() {{}}\nfn main() {{}}"), - ("fn-no-body", "pub trait T {{ {A} fn m(&self); }}\nfn main() {{}}"), - ("foreign-fn", 'unsafe extern "C" {{ {A} fn ext(); }}\nfn main() {{}}'), - ("param", "pub fn f({A} x: u32) -> u32 {{ x }}\nfn main() {{}}"), - ("param-no-body", "pub trait T {{ fn m(&self, {A} x: u32); }}\nfn main() {{}}"), - ("fn-ptr-param", "pub type F = fn({A} u32);\nfn main() {{}}"), - ("struct", "{A}\npub struct S;\nfn main() {{}}"), - ("field", "pub struct S {{ {A} pub x: u32 }}\nfn main() {{}}"), - ("impl", "pub struct S;\n{A}\nimpl S {{}}\nfn main() {{}}"), - ("trait", "{A}\npub trait T {{}}\nfn main() {{}}"), - ("mod", "{A}\npub mod m {{}}\nfn main() {{}}"), - ("closure", "fn main() {{ let _c = {A} || (); }}"), - ("statement", "fn main() {{ {A} let _x = 1; }}"), - ("crate", "#![{AI}]\nfn main() {{}}"), -] - - -def compile_text(text, feature=None, crate_type="bin"): - with tempfile.TemporaryDirectory(dir=WORK) as d: - f = Path(d) / "t.rs" - f.write_text((f"#![feature({feature})]\n" if feature else "") + text) - env = dict(ENV, RUSTC_BOOTSTRAP="1") if feature else ENV - try: - r = subprocess.run([args.rustc, str(f), "--edition", "2021", "--emit=metadata", "--crate-type", - crate_type, "-o", str(Path(d) / "x")], capture_output=True, text=True, - timeout=60, cwd=d, env=env) - except subprocess.TimeoutExpired: - return "timeout", "" - if "internal compiler error" in r.stderr or "panicked" in r.stderr: - return "ice", r.stderr - return ("ok" if r.returncode == 0 else "error"), r.stderr - - -def classify(status, stderr, item=None): - # `#[feature]` outside the crate root does nothing; rustc warns that it belongs at the root. - if status == "ok" and item == "feature" and "crate-level attribute" in stderr: - return "gated" - if status == "ok": - return "accepted" - if status == "ice": - return "ice" - return "gated" if GATE.search(stderr) else "no-gate-message" - - -def attributes(): - text = (RUST / "compiler/rustc_feature/src/builtin_attrs.rs").read_text() - names = sorted(set(re.findall(r"sym::([a-z_0-9]+)", text[text.index("Unstable attributes:"):]))) - jobs = [] - for name in names: - arg = ATTR_ARGS.get(name, "") - for pos, template in POSITIONS: - jobs.append((name, pos, template.replace("{AI}", f"{name}{arg}").replace("{A}", f"#[{name}{arg}]") - .replace("{{", "{").replace("}}", "}"))) - - def run(job): - name, pos, prog = job - status, err = compile_text(prog) - return {"kind": "attribute", "item": name, "position": pos, "result": classify(status, err, name), - "first": next((l for l in err.splitlines() if l.startswith("error")), "")[:200]} - with ThreadPoolExecutor(args.jobs) as ex: - return list(ex.map(run, jobs)) - - -def library_items(): - items = [] - for krate in ("core", "alloc", "std"): - root = RUST / "library" / krate / "src" - for f in root.rglob("*.rs"): - rel = f.relative_to(root).with_suffix("") - parts = [x for x in rel.parts if x not in ("lib", "mod")] - module = "::".join([krate] + parts) - lines = f.read_text(errors="replace").splitlines() - for i, line in enumerate(lines): - m = re.match(r'#\[unstable\(feature = "([a-z_0-9]+)"', line) - if not m: - continue - for nxt in lines[i + 1:i + 6]: - if nxt.startswith("#"): - continue - d = re.match(r"pub (?:const |unsafe |auto |extern \"C\" )*(struct|enum|trait|union|type|fn|const|static|macro) ([A-Za-z_][A-Za-z_0-9]*)(<)?", nxt) - if d: - items.append({"crate": krate, "path": f"{module}::{d.group(2)}", "kind": d.group(1), - "feature": m.group(1), "generic": bool(d.group(3)), - "file": str(f.relative_to(RUST))}) - break - return items - - -def library(): - items = library_items() - - def run(item): - path, feature, kind = item["path"], item["feature"], item["kind"] - # The path must resolve with the feature on, or the item is not where its file says. - status, _ = compile_text(f"#[allow(unused_imports)] use {path};\nfn main() {{}}", feature) - if status != "ok": - return [{"kind": "library", "item": path, "position": "path", "result": "skipped: path"}] - parent, name = path.rsplit("::", 1) - progs = [("use", f"#[allow(unused_imports)] use {path};\nfn main() {{}}"), - ("use-as", f"#[allow(unused_imports)] use {path} as Renamed;\nfn main() {{}}"), - ("glob", f"#[allow(unused_imports)] use {parent}::*;\n#[allow(unused_imports)] use self::{name} as _;\nfn main() {{}}")] - if kind == "trait" and not item["generic"]: - progs.append(("impl", f"struct L;\nimpl {path} for L {{}}\nfn main() {{}}")) - if kind == "fn" and not item["generic"]: - progs.append(("value", f"fn main() {{ let _f = {path}; }}")) - if kind in ("struct", "enum", "union", "type") and not item["generic"]: - progs.append(("type", f"pub fn g(_: Option<&{path}>) {{}}\nfn main() {{}}")) - out = [] - for pos, prog in progs: - status, err = compile_text(prog) - out.append({"kind": "library", "item": path, "feature": feature, "position": pos, - "result": classify(status, err), - "first": next((l for l in err.splitlines() if l.startswith("error")), "")[:200]}) - return out - with ThreadPoolExecutor(args.jobs) as ex: - return [r for rs in ex.map(run, items) for r in rs] - - -def main(): - results = [] - if args.only in (None, "attributes"): - results += attributes() - if args.only in (None, "library"): - results += library() - (WORK / "results.json").write_text(json.dumps(results, indent=0)) - from collections import Counter - print(Counter((r["kind"], r["result"]) for r in results)) - for r in results: - if r["result"] in ("accepted", "ice"): - print(f"{r['result'].upper():9} {r['kind']:9} {r['item']:45} {r['position']}") - - -main() diff --git a/rustc/instr-check.py b/rustc/instr-check.py deleted file mode 100644 index f4521ce..0000000 --- a/rustc/instr-check.py +++ /dev/null @@ -1,162 +0,0 @@ -#!/usr/bin/env python3 -"""Instrumentation round trip: instrumented programs must behave as uninstrumented ones and write -profiles that LLVM's tools accept. - -For each runnable UI test (run-pass), with a toolchain that ships the profiler runtime and -llvm-tools (the pinned nightly): - - pgo build with `-Cprofile-generate`, run, `llvm-profdata merge` the raw profile, rebuild - with `-Cprofile-use`, run again - coverage build with `-Cinstrument-coverage`, run, merge, `llvm-cov export` the binary - -Findings: an instrumented or profile-guided program whose exit status or stdout differs from the -plain build; no raw profile written; `llvm-profdata` or `llvm-cov` failing or crashing, or -warning about corrupt or malformed data; the compiler crashing on `-Cprofile-use`. - - rustc/instr-check.py --toolchain nightly-2026-10-06 --tests /tests/ui --work - [--only ] [--limit N] [--jobs 8] [--pause-on-finding] [--recheck] -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -p = argparse.ArgumentParser() -p.add_argument("--toolchain", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--limit", type=int) -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -TC = Path.home() / f".rustup/toolchains/{args.toolchain}-x86_64-unknown-linux-gnu" -RUSTC = TC / "bin/rustc" -TOOLS = TC / "lib/rustlib/x86_64-unknown-linux-gnu/bin" -OWN = re.compile(r"profile|instrument-coverage|coverage-options|-O$|opt-level|panic=|prefer-dynamic|" - r"codegen-backend|-Clto|lto=|no-prepopulate") -BAD = re.compile(r"corrupt|malformed|invalid|truncated|failed to|error", re.I) - - -def tool(name, *a, cwd=None): - try: - r = subprocess.run([str(TOOLS / name), *a], capture_output=True, text=True, timeout=120, cwd=cwd) - return r.returncode, r.stderr + r.stdout[-200:] - except subprocess.TimeoutExpired: - return "timeout", "" - - -def observe(binary, env=None): - try: - r = subprocess.run([str(binary)], capture_output=True, timeout=30, cwd=binary.parent, - stdin=subprocess.DEVNULL, env=dict(os.environ, RUST_BACKTRACE="0", **(env or {}))) - except subprocess.TimeoutExpired: - return ("timeout", "") - code = r.returncode if r.returncode >= 0 else f"signal {-r.returncode}" - return (code, r.stdout.decode(errors="replace")) - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel} - found = [] - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - status, _, plain = uitest.compile(RUSTC, path.resolve(), d / "plain", flags, edition, ["-Copt-level=1"]) - if plain is None: - record["skip"] = f"plain build {status}" - return record, [] - base = observe(plain) - if base != observe(plain): - record["skip"] = "nondeterministic" - return record, [] - # PGO: generate, merge, use. - raw = d / "pgo-raw" - status, err, gen = uitest.compile(RUSTC, path.resolve(), d / "gen", flags, edition, - ["-Copt-level=1", f"-Cprofile-generate={raw}"]) - if gen is None: - found.append({"what": f"profile-generate build {status}", "stderr": err[-1500:]}) - else: - got = observe(gen) - if got != base: - found.append({"what": "profile-generate changes behavior", "base": base, "got": got}) - profraws = list(raw.glob("*.profraw")) if raw.exists() else [] - if not profraws and got[0] == 0: - found.append({"what": "no raw profile written"}) - elif profraws: - merged = d / "merged.profdata" - code, out = tool("llvm-profdata", "merge", "-o", str(merged), *map(str, profraws)) - if code != 0 or BAD.search(out): - found.append({"what": "llvm-profdata merge", "code": code, "out": out[-1500:]}) - else: - status, err, use = uitest.compile(RUSTC, path.resolve(), d / "use", flags, edition, - ["-Copt-level=2", f"-Cprofile-use={merged}"]) - if status == "ice": - found.append({"what": "profile-use ICE", "stderr": err[-2000:]}) - elif use is None: - found.append({"what": f"profile-use build {status}", "stderr": err[-1500:]}) - elif observe(use) != base: - found.append({"what": "profile-use changes behavior", "base": base, "got": observe(use)}) - # Coverage: instrument, merge, export. - status, err, cov = uitest.compile(RUSTC, path.resolve(), d / "cov", flags, edition, ["-Cinstrument-coverage"]) - if cov is None: - found.append({"what": f"instrument-coverage build {status}", "stderr": err[-1500:]}) - else: - craw = d / "cov-raw" - craw.mkdir() - got = observe(cov, {"LLVM_PROFILE_FILE": str(craw / "c-%p.profraw")}) - if got != base: - found.append({"what": "instrument-coverage changes behavior", "base": base, "got": got}) - profraws = list(craw.glob("*.profraw")) - if profraws: - merged = d / "cov.profdata" - code, out = tool("llvm-profdata", "merge", "-sparse", "-o", str(merged), *map(str, profraws)) - if code != 0 or BAD.search(out): - found.append({"what": "llvm-profdata merge (coverage)", "code": code, "out": out[-1500:]}) - else: - code, out = tool("llvm-cov", "export", "-summary-only", f"-instr-profile={merged}", str(cov)) - if code != 0: - found.append({"what": "llvm-cov export", "code": code, "out": out[-1500:]}) - elif got[0] == 0: - found.append({"what": "no coverage profile written"}) - record["found"] = [f["what"] for f in found] - if found: - out = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - shutil.copy(path, out / path.name) - (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, - "found": found}, indent=1, default=str)) - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, ("run-pass",), - lambda text, flags: any(OWN.search(f) for f in flags)): - rel = str(path.relative_to(args.tests)) - if (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - if args.limit: - todo = todo[:args.limit] - print(f"{len(todo)} tests", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/miri-diff.py b/rustc/miri-diff.py deleted file mode 100644 index a0bf77c..0000000 --- a/rustc/miri-diff.py +++ /dev/null @@ -1,168 +0,0 @@ -#!/usr/bin/env python3 -"""Miri differential: an accepted program must be free of undefined behavior, MIR optimizations -must not introduce any, and the compiled program must do what Miri says it does. - -For each runnable UI test (run-pass, run-fail), interprets it with Miri three times: without MIR -optimizations (Miri's default), at `-Zmir-opt-level=2` (what `-O` runs) and at -`-Zmir-opt-level=4` (every MIR pass), and builds and runs it with the compiler under test, -unoptimized. Findings: - - ub Miri reports undefined behavior at mir-opt-level 0 in an accepted program without - `unsafe` code: the compiler accepted something unsound (UB in a test with its own - unsafe code is noted, not reported) - ub-opt undefined behavior only after MIR optimization: a MIR pass broke the program - opt-differs Miri's exit status or stdout changes with the MIR optimization level - native the compiled program's exit status or stdout differs from Miri's - -Overflow checks and debug assertions are on everywhere. Tests Miri cannot run (foreign -functions, inline assembly, unsupported operations) or that time out are skipped. - - rustc/miri-diff.py --rustc --tests /tests/ui --work [--only ] - [--known ] [--jobs 8] [--timeout 120] [--pause-on-finding] [--recheck] - -Writes /results.jsonl and /findings//. -""" - -import argparse -import json -import re -import shutil -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -FIXED = ["-Coverflow-checks=on", "-Cdebug-assertions=on"] -# Miri without preemption: threads switch only where they block, the same at every MIR level. -MIRI_FLAGS = ["-Zmiri-preemption-rate=0"] -# Tests asserting what Rust leaves unspecified, which Miri varies on purpose: function pointer -# equality, the addresses of zero-sized values, stack addresses, function alignment. -UNSPECIFIED = {"consts/const-extern-function.rs", "consts/zst_no_llvm_alloc.rs", - "layout/null-pointer-optimization.rs", "mir/mir_misc_casts.rs", "mir/mir_coercions.rs", - "extern/extern-compare-with-return-type.rs", "fn/fn-ptr-trait-run.rs", "mir/mir_raw_fat_ptr.rs", - "codegen/StackColoring-not-blowup-stack-issue-40883.rs", "attributes/fn-align-dyn.rs", - # Miri calls .init_array functions without glibc's (argc, argv, envp). - "runtime/stdout-before-main.rs"} -THREADS = re.compile(r"thread::(spawn|scope)|std::sync::mpsc|\bspawn\(") -LEVELS = {"miri0": [], "miri2": ["-Zmir-opt-level=2"], "miri4": ["-Zmir-opt-level=4"]} -OWN = re.compile(r"^-O$|opt-level|overflow-checks|debug-assertions|codegen-backend|mir-enable-passes|" - r"panic=|-Cpanic|prefer-dynamic|-Zbuild-std|-Clink|-Ctarget") -# Code Miri cannot interpret: skip without trying. -NOT_FOR_MIRI = re.compile(r"\basm!|global_asm!|naked_asm!|extern\s+\"C\"\s*\{|#\[link\(|std::process::Command|" - r"\bfork\b|libc::|dlopen|std::os::unix::process") - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--timeout", type=int, default=120) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def clean(text, path): - text = re.sub(r"(thread '[^']*') \(\d+\)", r"\1", text) - text = re.sub(r"\S*/lib/rustlib/src/rust/library/", "library/", text) - # argv[0]: the source file under Miri, the binary natively. - text = text.replace(str(path.resolve()), "") - text = re.sub(r"\S*/native/prog\b", "", text) - # The test harness: timings, and result lines in completion order. - text = re.sub(r"finished in \d+\.\d+s", "finished in …s", text) - lines = text.split("\n") - results = iter(sorted(l for l in lines if l.startswith("test ") and " ... " in l)) - return "\n".join(next(results) if (l.startswith("test ") and " ... " in l) else l for l in lines) - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel, "kind": kind} - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - status, _, binary = uitest.compile(args.rustc, path.resolve(), d / "native", flags, edition, - FIXED + ["-Copt-level=0"]) - if binary is None: - record["skip"] = f"native build {status}" - return record, [] - runs = [uitest.run(binary) for _ in range(3)] - outs = {(c, clean(o.decode(errors="replace"), path)) for c, o, _ in runs} - if len(outs) > 1: - record["skip"] = "native run is nondeterministic" - return record, [] - code, out = outs.pop() - native = {"exit": code, "stdout": out} - miri = {} - for name, extra in LEVELS.items(): - m = uitest.miri(path.resolve(), flags, edition, FIXED + MIRI_FLAGS + extra, timeout=args.timeout, cwd=d) - m["stdout"] = clean(m["stdout"], path) - miri[name] = m - if name == "miri0" and m["status"] in ("unsupported", "error", "timeout"): - record["skip"] = f"miri {m['status']}" - record["why"] = m["stderr"][-300:] - return record, [] - record["miri"] = {k: v["status"] for k, v in miri.items()} - found = [] - # UB in a program without `unsafe` code can only be the compiler's; in a test with its - # own unsafe code it is most likely the test's (noted, not a finding). - safe = "unsafe" not in path.read_text(errors="replace") - if miri["miri0"]["status"] == "ub" and safe and rel not in UNSPECIFIED: - found.append({"what": "ub (safe code)", "stderr": miri["miri0"]["stderr"][-3000:]}) - elif miri["miri0"]["status"] == "ub": - record["note"] = "ub in a test with unsafe code" - threaded = THREADS.search(path.read_text(errors="replace")) - for name in ("miri2", "miri4"): - m = miri[name] - if m["status"] == "ub" and miri["miri0"]["status"] != "ub": - found.append({"what": f"ub-opt ({name})", "stderr": m["stderr"][-3000:]}) - elif m["status"] == "ice": - found.append({"what": f"ice ({name})", "stderr": m["stderr"][-3000:]}) - elif (m["status"] == "ok" and miri["miri0"]["status"] == "ok" and not threaded - and (m["exit"], m["stdout"]) != (miri["miri0"]["exit"], miri["miri0"]["stdout"])): - found.append({"what": f"opt-differs ({name})", "miri0": miri["miri0"]["stdout"][-1500:], - "got": m["stdout"][-1500:], "exits": [miri["miri0"]["exit"], m["exit"]]}) - m0 = miri["miri0"] - # Native threads race; Miri's do not without preemption: no comparison for threaded tests. - if m0["status"] == "ok" and not threaded and rel not in UNSPECIFIED: - # Miri exits 1 on a panic that reaches main, native code 101. - exit_m = 101 if m0["exit"] == 1 and "panicked" in m0["stderr"] else m0["exit"] - if (exit_m, m0["stdout"]) != (native["exit"], native["stdout"]): - found.append({"what": "native", "miri": [m0["exit"], m0["stdout"][-1500:]], - "native": [native["exit"], native["stdout"][-1500:]]}) - if found: - outdir = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(outdir, ignore_errors=True) - outdir.mkdir(parents=True) - shutil.copy(path, outdir / path.name) - (outdir / "finding.json").write_text(json.dumps( - {"test": rel, "flags": flags, "edition": edition, "fixed": FIXED, "found": found}, indent=1)) - record["found"] = [f["what"] for f in found] - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests( - args.tests, ("run-pass", "run-fail"), - lambda text, flags: any(OWN.search(f) for f in flags) or NOT_FOR_MIRI.search(text) - # Compile-time output (trace_macros, log_syntax) would land in Miri's stdout. - or "trace_macros" in text or "log_syntax" in text): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/opt-diff.py b/rustc/opt-diff.py deleted file mode 100644 index e9910da..0000000 --- a/rustc/opt-diff.py +++ /dev/null @@ -1,185 +0,0 @@ -#!/usr/bin/env python3 -"""Optimization differential: a program's behavior must not depend on how it was optimized. - -Builds each runnable UI test (run-pass, run-fail) under a set of configurations (optimization -levels, MIR optimization levels, LTO, target CPU, the Cranelift backend) and runs it. Overflow -checks and debug assertions are fixed across configurations, so the program's semantics are the -same in all of them. Compared with the unoptimized baseline (`-Copt-level=0 -Zmir-opt-level=0`): - - run exit status and stdout (stderr too, with panic locations kept, backtrace hints dropped) - build a configuration fails to build, or crashes the compiler, where the baseline builds - -A baseline whose output varies between two runs is nondeterministic and skipped; a difference -is confirmed by running both binaries again before it counts. - - rustc/opt-diff.py --rustc --tests /tests/ui --work - [--cranelift ] [--configs O3,O3-lto] [--only ] - [--known ] [--jobs 8] [--pause-on-finding] [--recheck] - -Writes /results.jsonl, and /findings// with the source, the configurations' -argv and outputs. With --pause-on-finding, stops starting new tests at the first finding and -exits 3 (the frontier loop: docs/hunt.md). --recheck runs only the tests with findings. -""" - -import argparse -import json -import re -import shutil -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -FIXED = ["-Coverflow-checks=on", "-Cdebug-assertions=on", "-Cpanic=unwind", "-Cdebuginfo=0"] -CONFIGS = { - "base": ["-Copt-level=0", "-Zmir-opt-level=0"], - "O0": ["-Copt-level=0"], - "O0-mir4": ["-Copt-level=0", "-Zmir-opt-level=4"], - "O1": ["-Copt-level=1"], - "O2": ["-Copt-level=2"], - "O3": ["-Copt-level=3"], - "Os": ["-Copt-level=s"], - "Oz": ["-Copt-level=z"], - "O3-mir4": ["-Copt-level=3", "-Zmir-opt-level=4"], - "O3-lto": ["-Copt-level=3", "-Clto=fat", "-Ccodegen-units=1"], - "O2-cgu16": ["-Copt-level=2", "-Ccodegen-units=16"], - "O3-native": ["-Copt-level=3", "-Ctarget-cpu=native"], - "cranelift": ["-Copt-level=0", "-Zcodegen-backend=cranelift"], -} -# Tests whose outcome legitimately depends on optimization: unspecified behavior (whether two -# equal promoted constants share an address), stack usage, or a backend's documented gaps. -NOISE = { - "mir/mir_raw_fat_ptr.rs": {"cranelift"}, # compares the addresses of two `&0u8` - "codegen/StackColoring-not-blowup-stack-issue-40883.rs": {"O0-mir4", "O0"}, # stack usage - "attributes/fn-align-dyn.rs": {"cranelift"}, # Cranelift ignores #[align] on functions - "backtrace/backtrace.rs": {"cranelift"}, # Cranelift backtraces lack frames -} -# Tests that choose these themselves are left out: the configuration would contradict them. -OWN = re.compile(r"^-O$|opt-level|mir-opt-level|overflow-checks|debug-assertions|codegen-backend|" - r"mir-enable-passes|^-Clto|lto=|target-cpu|panic=|-Cpanic|prefer-dynamic|-Zbuild-std") - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--cranelift", help="a rustc with the cranelift backend (the pinned nightly)") -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--configs", help="comma-separated subset (base is always built)") -p.add_argument("--only", help="tests whose path contains this") -p.add_argument("--known", help="file of test paths to leave out (known findings)") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) -(WORK / "scratch").mkdir(exist_ok=True) -configs = {k: v for k, v in CONFIGS.items() - if k == "base" or not args.configs or k in args.configs.split(",")} -if not args.cranelift: - configs.pop("cranelift", None) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def normalized(stderr): - lines = [l for l in stderr.decode(errors="replace").splitlines() - if not l.startswith(("note: run with `RUST_BACKTRACE", "note: Some details are omitted"))] - text = "\n".join(lines) - # Panic messages name the thread with its OS id: `thread 'main' (909942) panicked`. - text = re.sub(r"(thread '[^']*') \(\d+\)", r"\1", text) - # Toolchains print std's and dependencies' paths differently: in full (rust-src installed), - # remapped to /rustc//, or relative. - text = re.sub(r"\S*/lib/rustlib/src/rust/library/|/rustc/[0-9a-f]+/library/", "library/", text) - return re.sub(r"\S*/registry/(src/)?[^/\s]+/([^/\s]+-\d[^/\s]*)/", r"/\2/", text) - - -def observe(binary): - # Every configuration's program runs from the same path: some tests print argv[0]. - fixed = binary.parent.parent / "run" / "prog" - fixed.parent.mkdir(exist_ok=True) - shutil.copy2(binary, fixed) - code, out, err = uitest.run(fixed) - stdout = out.decode(errors="replace") - # The test harness prints how long tests took, and runs tests on several threads: their - # result lines come in any order. - stdout = re.sub(r"finished in \d+\.\d+s", "finished in …s", stdout) - lines = stdout.split("\n") - results = sorted(l for l in lines if l.startswith("test ") and " ... " in l) - it = iter(results) - stdout = "\n".join(next(it) if (l.startswith("test ") and " ... " in l) else l for l in lines) - return {"exit": code, "stdout": stdout, "stderr": normalized(err)} - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - built, runs = {}, {} - text = path.read_text(errors="replace") - for name, cfg in configs.items(): - # Cranelift does not unwind on this target yet (catch_unwind catches nothing). - if name == "cranelift" and ("catch_unwind" in text or "needs-unwind" in text): - continue - rustc = args.cranelift if name == "cranelift" else args.rustc - status, stderr, binary = uitest.compile(rustc, path.resolve(), d / name, flags, edition, FIXED + cfg) - built[name] = {"status": status, "stderr": stderr[-2000:]} - if binary: - runs[name] = observe(binary) - record = {"test": rel, "kind": kind, "build": {k: v["status"] for k, v in built.items()}} - if built["base"]["status"] != "ok": - record["skip"] = "baseline does not build" - return record, [] - again = observe(d / "base" / "prog") - if again != runs["base"]: - record["skip"] = "baseline is nondeterministic" - return record, [] - found = [] - for name in configs: - if name == "base" or name not in built or name in NOISE.get(rel, ()): - continue - b = built[name]["status"] - if b != "ok": - # The Cranelift backend has documented gaps (tail calls, some linkages and SIMD - # intrinsics), which it reports as errors or as panics inside itself. - if name == "cranelift" and (b == "error" or "rustc_codegen_cranelift" in built[name]["stderr"]): - continue - found.append({"config": name, "what": f"build {b}", "stderr": built[name]["stderr"]}) - continue - # Cranelift cannot unwind on this target yet: a panic aborts. - if name == "cranelift" and "failed to initiate panic" in runs[name]["stderr"]: - continue - if runs[name] != runs["base"]: - retry = observe(d / name / "prog") - if retry != runs["base"] and retry == runs[name]: - diff = [k for k in ("exit", "stdout", "stderr") if runs[name][k] != runs["base"][k]] - found.append({"config": name, "what": "run differs: " + ",".join(diff), - "base": runs["base"], "got": runs[name]}) - if found: - out = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - shutil.copy(path, out / path.name) - (out / "finding.json").write_text(json.dumps({ - "test": rel, "flags": flags, "edition": edition, "fixed": FIXED, - "configs": {f["config"]: configs[f["config"]] for f in found}, "found": found, - "base": runs["base"]}, indent=1)) - record["found"] = [f"{f['config']}: {f['what']}" for f in found] - return record, found - - -def main(): - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, ("run-pass", "run-fail"), - lambda text, flags: any(OWN.search(f) for f in flags)): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (args.recheck and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests, configurations: {', '.join(configs)}", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/release-diff.py b/rustc/release-diff.py deleted file mode 100644 index 87d2ebd..0000000 --- a/rustc/release-diff.py +++ /dev/null @@ -1,124 +0,0 @@ -#!/usr/bin/env python3 -"""Release-to-release: code that one toolchain accepts the next must accept too, in comparable -time. - -Runs `cargo check --locked` on each repository of a corpus of real crates with two toolchains -(an older and a newer rustup toolchain, or a local rustc via a rustup-linked name), each in its -own target directory, deleted afterwards. Compared: - - regression the older toolchain checks the repository, the newer one does not (the new - error codes and the first error are recorded); noted instead when the failing crate - enables unstable features (`#![feature]`, often only when it detects a nightly) - fixed the other way round (reported, not a finding) - slower the newer toolchain takes more than --slower times as long (both succeed) - ice the newer toolchain crashes - -Dependencies are fetched first (`cargo fetch --locked`), so the timed runs are offline. - - rustc/release-diff.py --corpus --old --new - --work [--only ] [--jobs 2] [--slower 1.5] [--timeout 1800] - -Writes /results.jsonl and prints regressions, crashes and slowdowns. -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -import sys -import time -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--corpus", required=True) -p.add_argument("--old", required=True) -p.add_argument("--new", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--jobs", type=int, default=2) -p.add_argument("--slower", type=float, default=1.5) -p.add_argument("--timeout", type=int, default=1800) -args = p.parse_args() -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) - - -def check(repo, toolchain): - target = WORK / "target" / f"{repo.name}-{toolchain}" - shutil.rmtree(target, ignore_errors=True) - env = dict(os.environ, CARGO_TARGET_DIR=str(target), CARGO_TERM_COLOR="never", - CARGO_INCREMENTAL="0", RUSTFLAGS="--cap-lints=warn") - env.pop("RUSTC_WRAPPER", None) - start = time.time() - try: - r = subprocess.run(["cargo", f"+{toolchain}", "check", "--locked", "--offline", "--workspace", - "--message-format=short"], cwd=repo, env=env, capture_output=True, text=True, - timeout=args.timeout) - code, err = r.returncode, r.stderr - except subprocess.TimeoutExpired: - code, err = "timeout", "" - seconds = round(time.time() - start, 1) - shutil.rmtree(target, ignore_errors=True) - ice = "internal compiler error" in err or "the compiler unexpectedly panicked" in err - codes = sorted(set(re.findall(r"error\[(E\d{4})\]", err))) - first = next((l for l in err.splitlines() if re.search(r"\berror(\[E\d+\])?:", l)), "") - return {"code": code, "seconds": seconds, "ice": ice, "codes": codes, "first": first[:300], - "tail": err[-3000:] if code != 0 else ""} - - -def uses_unstable(first_error): - """The crate a first error points into, if its source enables `#![feature(...)]`.""" - m = re.search(r"(/\S*?/registry/src/[^/]+/[^/]+|/\S+?)/src/", first_error) - if not m: - return None - root = Path(m.group(1)) - for f in list((root / "src").glob("lib.rs")) + list((root / "src").glob("main.rs")): - text = f.read_text(errors="replace") - if re.search(r"#!\[(cfg_attr\([^]]*)?feature\(", text): - return root.name - return None - - -def one(repo): - fetch = subprocess.run(["cargo", f"+{args.new}", "fetch", "--locked"], cwd=repo, capture_output=True, - text=True, timeout=1800) - if fetch.returncode != 0: - return {"repo": repo.name, "skip": "fetch failed", "why": fetch.stderr[-400:]} - old = check(repo, args.old) - new = check(repo, args.new) - rec = {"repo": repo.name, "old": old, "new": new, "found": []} - if old["code"] == 0 and new["code"] != 0: - unstable = uses_unstable(new["first"]) - if unstable and not new["ice"]: - # A crate that turns on unstable features when it detects a nightly compiler breaks - # when they change: expected between nightlies, noted rather than reported. - rec["notes"] = [f"regression in a crate using unstable features ({unstable})"] - else: - rec["found"].append("ice" if new["ice"] else "regression") - elif old["code"] != 0 and new["code"] == 0: - rec["notes"] = ["fixed"] - elif old["code"] == 0 and new["code"] == 0 and old["seconds"] > 5 and new["seconds"] > args.slower * old["seconds"]: - rec["found"].append(f"slower: {old['seconds']}s -> {new['seconds']}s") - if new["ice"] and "ice" not in rec["found"]: - rec["found"].append("ice") - return rec - - -def main(): - repos = sorted(p for p in Path(args.corpus).iterdir() if (p / "Cargo.toml").exists() - and (not args.only or args.only in p.name)) - print(f"{len(repos)} repositories, {args.old} -> {args.new}", flush=True) - with ThreadPoolExecutor(args.jobs) as ex, (WORK / "results.jsonl").open("a") as out: - for rec in ex.map(one, repos): - out.write(json.dumps(rec) + "\n") - out.flush() - tag = rec.get("skip") or ", ".join(rec.get("found", [])) or "same" - old, new = rec.get("old", {}), rec.get("new", {}) - print(f"{rec['repo']:45} {tag:30} old {old.get('code')} {old.get('seconds')}s " - f"new {new.get('code')} {new.get('seconds')}s {new.get('first', '')[:80]}", flush=True) - - -main() diff --git a/rustc/repro-diff.py b/rustc/repro-diff.py deleted file mode 100644 index 1770b67..0000000 --- a/rustc/repro-diff.py +++ /dev/null @@ -1,146 +0,0 @@ -#!/usr/bin/env python3 -"""Determinism: what rustc writes must depend only on its inputs and options. - -Builds each standalone UI test that compiles (build-pass, run-pass, check-pass) several times -and compares the outputs (`.rmeta`, `.rlib` normalized as artifacts.py does, executables) with -the first build: - - repeat the same build again, in the same directory - path the same build in another directory, both with `--remap-path-prefix` to one name: - what is left of the directory is a path leaking past the remapping - threads `-Zthreads=8` (the parallel front end; tests marked `ignore-parallel-frontend` skip - it); known: async fns (rust-lang/rust#162202), RPIT and impl Trait in traits (#163878) - decoy a `-L` directory holding an unrelated library whose name starts with the crate's - name (rust-lang/rust#159677's shape) - - rustc/repro-diff.py --rustc --tests /tests/ui --work [--variants repeat,path] - [--only ] [--known ] [--jobs 8] [--pause-on-finding] [--recheck] -""" - -import argparse -import hashlib -import json -import os -import re -import shutil -import subprocess -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import artifacts # noqa: E402 -import uitest # noqa: E402 - -VARIANTS = ["repeat", "path", "threads", "decoy"] - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--variants") -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -variants = args.variants.split(",") if args.variants else VARIANTS -known = set(Path(args.known).read_text().split()) if args.known else set() -OWN = re.compile(r"threads|remap-path|-o\b|--out-dir|emit|crate-name|extern|-L\b|-Cincremental") - - -def build(src_dir, name, flags, edition, kind, extra): - """Compile `src_dir/name`; {output file: digest}, or None if it fails.""" - out = src_dir / "out" - shutil.rmtree(out, ignore_errors=True) - out.mkdir() - emit = "--emit=metadata" if kind == "check-pass" else "--emit=link,metadata" - argv = [args.rustc, name, "--edition", edition or "2015", emit, "--out-dir", "out", "--crate-name", "t", - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", - f"--remap-path-prefix={src_dir}=/src", *flags, *extra] - try: - r = subprocess.run(argv, capture_output=True, timeout=300, cwd=src_dir, - env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) - except subprocess.TimeoutExpired: - return None - if r.returncode != 0: - return None - files = {} - for f in sorted(out.iterdir()): - if f.suffix == ".rlib": - files[f.name] = artifacts.normalized_rlib(f) - elif f.is_file(): - files[f.name] = hashlib.sha256(f.read_bytes()).hexdigest() - return files - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel} - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - a, b = d / "a", d / "elsewhere-b" - for x in (a, b): - x.mkdir() - shutil.copy(path, x / path.name) - base = build(a, path.name, flags, edition, kind, []) - if base is None: - record["skip"] = "does not build" - return record, [] - found = [] - text = path.read_text(errors="replace") - for v in variants: - # Tests the parallel front end is known not to handle say so. - if v == "threads" and re.search(r"^//@\s*ignore-parallel-frontend", text, re.M): - continue - if v == "repeat": - got = build(a, path.name, flags, edition, kind, []) - elif v == "path": - got = build(b, path.name, flags, edition, kind, []) - elif v == "threads": - got = build(a, path.name, flags, edition, kind, ["-Zthreads=8"]) - elif v == "decoy": - decoy = d / "decoy" - decoy.mkdir(exist_ok=True) - # An unrelated library sharing the crate's name as a prefix. - (decoy / "libtother.rlib").write_bytes(b"!\n") - (decoy / "libt-0123456789abcdef.rmeta").write_bytes(b"rust\0\0\0\0") - got = build(a, path.name, flags, edition, kind, ["-L", str(decoy)]) - if got is None: - found.append({"variant": v, "what": "does not build"}) - elif got != base: - differ = sorted(k for k in set(got) | set(base) if got.get(k) != base.get(k)) - found.append({"variant": v, "what": "outputs differ: " + ", ".join(differ)}) - # A difference also in `repeat` is nondeterminism of the build itself: report only that. - if any(f["variant"] == "repeat" for f in found): - found = [f for f in found if f["variant"] == "repeat"] - record["found"] = [f"{f['variant']}: {f['what']}" for f in found] - if found: - out = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - shutil.copy(path, out / path.name) - (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, - "found": found}, indent=1)) - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, ("build-pass", "run-pass", "check-pass"), - lambda text, flags: any(OWN.search(f) for f in flags)): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests, variants: {', '.join(variants)}", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/rewrite-diff.py b/rustc/rewrite-diff.py deleted file mode 100644 index f87ad5a..0000000 --- a/rustc/rewrite-diff.py +++ /dev/null @@ -1,176 +0,0 @@ -#!/usr/bin/env python3 -"""Equivalent rewrites: rewriting a program into an equivalent one must not change its verdict. - -Each standalone UI test is printed back unchanged (`mirth-rewrite identity`: the baseline, since -printing drops comments and moves lines) and rewritten by each of mirth-rewrite's rewrites -(generic-wrap, alias, reorder, unused). Each version is compiled the way the test's headers say -(metadata for check tests, a full build for build and run tests), and compared with the -baseline: - - verdict accepted against rejected, or a crash on one side only (a finding) - codes both rejected with different sets of error codes (a finding for reorder and unused, - which change nothing a diagnostic could depend on; noted for the others) - -A test whose baseline differs from the original file's verdict is left out (the printer cannot -represent it faithfully). - - rustc/rewrite-diff.py --rustc --tests /tests/ui --work - [--rewrites generic-wrap,alias] [--only ] [--known ] [--jobs 8] - [--pause-on-finding] [--recheck] -""" - -import argparse -import json -import re -import shutil -import subprocess -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -REWRITER = Path(__file__).resolve().parent.parent / "target/release/mirth-rewrite" -REWRITES = ["generic-wrap", "alias", "reorder", "unused"] -# Error codes may differ legitimately for every rewrite (which error suppresses which depends on -# order): only a changed verdict is a finding. -STRICT = set() -# Differences that are resource limits or legitimate requirements of a generic context. -NOISE = { - ("consts/chained-constants-stackoverflow.rs", "reorder"), # 10,000 chained consts: query depth - ("consts/interior-mut-const-via-union.rs", "generic-wrap"), # finding 25 (docs/hunt.md) - # recursion_limit = "6": evaluation order nests the query stack one level deeper - ("traits/next-solver/overflow/dont-lower-depth-for-witness-and-rigid-opaque.rs", "reorder"), - ("imports/ambiguous-9.rs", "reorder"), # finding 28 - ("imports/ambiguous-14.rs", "reorder"), # finding 28 - ("imports/overwrite-different-ambig-2.rs", "reorder"), # finding 28 -} -NOT_MOVABLE = re.compile(r"^\s*(pub(\([^)]*\))?\s+)?mod\s+\w+\s*;|include(_str|_bytes)?!|#\[path|#!\[no_core\]", re.M) -# Item order matters to textual macro scoping: no reordering where macros are defined. -ORDER_MATTERS = re.compile(r"macro_rules!|#\[macro_use\]|macro\s+\w+") -KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--rewrites", help="comma-separated subset") -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -rewrites = args.rewrites.split(",") if args.rewrites else REWRITES -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def codes(stderr): - return sorted(set(re.findall(r"error\[(E\d{4})\]", stderr))) - - -def verdict(source, flags, edition, kind, out, has_main): - # A full build whenever there is a program: generic-wrap moves errors to monomorphization. - emit = "link" if has_main or kind not in ("check-pass", "check-fail", None) else "metadata" - # Lints capped: a rewrite may add or move a warning, and lint levels are not the subject. - status, stderr, _ = uitest.compile(args.rustc, source, out, flags, edition, ["--cap-lints=warn"], - timeout=120, emit=emit) - return {"status": status, "codes": codes(stderr), "stderr": stderr[-2500:]} - - -def rewrite(name, source, target): - r = subprocess.run([str(REWRITER), name, str(source)], capture_output=True, text=True, timeout=60) - if r.returncode != 0: - return False - target.write_text(r.stdout) - return True - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel, "kind": kind} - text = path.read_text(errors="replace") - # The rewritten file is compiled elsewhere: files it names by relative path are not there. - # Without `core`, a new trait or generic parameter does not compile. - if NOT_MOVABLE.search(text): - record["skip"] = "uses files by path or has no core" - return record, [] - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - base_src = d / "identity.rs" - if not rewrite("identity", path, base_src): - record["skip"] = "does not parse" - return record, [] - has_main = "fn main" in text - original = verdict(path.resolve(), flags, edition, kind, d / "original", has_main) - base = verdict(base_src, flags, edition, kind, d / "identity", has_main) - if (original["status"], original["codes"]) != (base["status"], base["codes"]): - record["skip"] = "printing changes the verdict" - return record, [] - record["base"] = base["status"] - found, notes, applied = [], [], [] - for name in rewrites: - if name == "reorder" and ORDER_MATTERS.search(text): - continue - if (rel, name) in NOISE: - continue - # generic_const_exprs requires `where` bounds in generic contexts that a concrete one - # does not. - if name == "generic-wrap" and "generic_const_exprs" in text: - continue - src = d / f"{name}.rs" - if not rewrite(name, path, src): - continue - applied.append(name) - v = verdict(src, flags, edition, kind, d / name, has_main) - if v["status"] != base["status"] and "timeout" not in (v["status"], base["status"]): - found.append({"rewrite": name, "what": f"verdict: {base['status']} -> {v['status']}", - "base_codes": base["codes"], "codes": v["codes"], "stderr": v["stderr"], - "base_stderr": base["stderr"]}) - elif v["status"] == base["status"] == "error" and v["codes"] != base["codes"]: - entry = {"rewrite": name, "what": f"codes: {base['codes']} -> {v['codes']}", - "stderr": v["stderr"], "base_stderr": base["stderr"]} - (found if name in STRICT else notes).append(entry) - if any(f["rewrite"] == name for f in found + notes): - shutil.copy(src, d / f"keep-{name}.rs") - record["applied"] = applied - record["found"] = [f"{f['rewrite']}: {f['what']}" for f in found] - record["notes"] = [f"{f['rewrite']}: {f['what']}" for f in notes] - if found or notes: - outdir = WORK / ("findings" if found else "notes") / rel.replace("/", "__") - shutil.rmtree(outdir, ignore_errors=True) - outdir.mkdir(parents=True) - shutil.copy(path, outdir / path.name) - shutil.copy(base_src, outdir / "identity.rs") - for f in found + notes: - shutil.copy(d / f"keep-{f['rewrite']}.rs", outdir / f"{f['rewrite']}.rs") - (outdir / "finding.json").write_text(json.dumps( - {"test": rel, "flags": flags, "edition": edition, "kind": kind, "found": found, "notes": notes}, - indent=1)) - return record, found - - -def main(): - global REWRITER - if not REWRITER.exists(): - sys.exit("build mirth-rewrite first: cargo build --release -p mirth-rewrite") - # A private copy: rebuilding mirth-rewrite must not change a sweep halfway. - shutil.copy2(REWRITER, WORK / "mirth-rewrite") - REWRITER = WORK / "mirth-rewrite" - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, KINDS): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests, rewrites: {', '.join(rewrites)}", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/scale-check.py b/rustc/scale-check.py deleted file mode 100644 index 924fc86..0000000 --- a/rustc/scale-check.py +++ /dev/null @@ -1,194 +0,0 @@ -#!/usr/bin/env python3 -"""Scaling and budgets: compile time, memory, future sizes and stack frames must grow at most -about linearly with the size of a program of a fixed shape. - -Each generator writes a program of size N for N in a doubling series. For each N the check records -the compiler's user CPU time and peak memory (wait4), and for some shapes a size the program itself -reports (`size_of_val` of a future) or the largest stack frame in the assembly. It fits the growth -exponent k in value ~ N^k from the largest sizes (log-log slope) and reports: - - superlinear k above --max-exponent (default 1.6) for time or memory, or above 1.3 for a size - timeout a compile over --timeout seconds - - rustc/scale-check.py --rustc --work [--only fields,enum] [--opt 0,2] - [--sizes 50,100,200,400,800] [--timeout 300] [--max-exponent 1.6] [--jobs 4] -""" - -import argparse -import json -import math -import os -import re -import subprocess -import tempfile -import time -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--opt", default="0,2") -p.add_argument("--sizes", default="50,100,200,400,800") -p.add_argument("--timeout", type=int, default=300) -p.add_argument("--max-exponent", type=float, default=1.6) -p.add_argument("--jobs", type=int, default=4) -args = p.parse_args() -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) - - -def g_fields(n): - fields = "\n".join(f" pub f{i}: u{8 << (i % 4)}," for i in range(n)) - return f"#[derive(Debug, Clone, PartialEq, Eq, Hash, Default, PartialOrd, Ord)]\npub struct S {{\n{fields}\n}}\nfn main() {{ let s = S::default(); println!(\"{{}}\", format!(\"{{:?}}\", s.clone()).len()); }}\n" - - -def g_enum(n): - variants = "\n".join(f" V{i}(u32)," for i in range(n)) - arms = "\n".join(f" E::V{i}(x) => x + {i}," for i in range(n)) - return f"#[derive(Debug, Clone, PartialEq)]\npub enum E {{\n{variants}\n}}\npub fn f(e: &E) -> u32 {{\n match *e {{\n{arms}\n }}\n}}\nfn main() {{ println!(\"{{}}\", f(&E::V0(1))); }}\n" - - -def g_nested_generic(n): - ty = "u8" - for _ in range(n): - ty = f"W<{ty}>" - return ("#[derive(Clone, Debug, Default)] pub struct W(T);\n" - "pub trait T { fn t(&self) -> usize; }\nimpl T for u8 { fn t(&self) -> usize { 1 } }\n" - "impl T for W { fn t(&self) -> usize { self.0.t() + 1 } }\n" - f"fn main() {{ let v: {ty} = Default::default(); println!(\"{{}}\", v.t()); }}\n") - - -def g_iter_chain(n): - chain = "".join(f".map(|x| x.wrapping_add({i}))" for i in range(n)) - return f"fn main() {{ let s: u64 = (0u64..10){chain}.sum(); println!(\"{{}}\", s); }}\n" - - -def g_async_forward(n): - fns = ["async fn f0(x: [u8; 64]) -> u8 { x[0] }"] - for i in range(1, n): - fns.append(f"async fn f{i}(x: [u8; 64]) -> u8 {{ f{i - 1}(x).await }}") - return ("\n".join(fns) + f"\nfn main() {{ let fut = f{n - 1}([1; 64]); " - "println!(\"SIZE {}\", std::mem::size_of_val(&fut)); }\n") - - -def g_seq_calls(n): - calls = "\n".join(f" let a{i} = big({i}); acc ^= a{i}[{i % 512}];" for i in range(n)) - return ("#[inline(never)] fn big(x: u64) -> [u64; 512] { [x; 512] }\n" - f"#[inline(never)] pub fn many() -> u64 {{\n let mut acc = 0u64;\n{calls}\n acc\n}}\n" - "fn main() { println!(\"{}\", many()); }\n") - - -def g_trait_impls(n): - impls = "\n".join(f"pub struct S{i}; impl Tr for S{i} {{ fn v(&self) -> u32 {{ {i} }} }}" for i in range(n)) - uses = " + ".join(f"S{i}.v()" for i in range(n)) - return f"pub trait Tr {{ fn v(&self) -> u32; }}\n{impls}\nfn main() {{ println!(\"{{}}\", {uses}); }}\n" - - -def g_nested_expr(n): - expr = "1u64" - for i in range(n): - expr = f"({expr} + {i % 7})" - return f"fn main() {{ let x = std::hint::black_box({expr}); println!(\"{{}}\", x); }}\n" - - -GENERATORS = { - "fields": (g_fields, None), "enum": (g_enum, None), "nested-generic": (g_nested_generic, None), - "iter-chain": (g_iter_chain, None), "async-forward": (g_async_forward, "run-size"), - "seq-calls": (g_seq_calls, "frame"), "trait-impls": (g_trait_impls, None), "nested-expr": (g_nested_expr, None), -} -# Shapes where a size of N cannot be compiled meaningfully past a bound (recursion limits). -CAP = {"nested-generic": 120, "nested-expr": 400, "async-forward": 200} - - -def measure(source, opt): - with tempfile.TemporaryDirectory(dir=WORK) as d: - d = Path(d) - (d / "m.rs").write_text(source) - argv = [args.rustc, "m.rs", "--edition", "2021", f"-Copt-level={opt}", "-o", "m", "--emit=link,asm"] - start = time.time() - proc = subprocess.Popen(argv, cwd=d, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE) - deadline = start + args.timeout - while True: - pid, status, usage = os.wait4(proc.pid, os.WNOHANG) - if pid: - break - if time.time() > deadline: - proc.kill() - os.wait4(proc.pid, 0) - return {"timeout": True} - time.sleep(0.05) - err = proc.stderr.read().decode(errors="replace") - if os.waitstatus_to_exitcode(status) != 0: - return {"error": err[-500:]} - out = {"user": usage.ru_utime, "rss_kb": usage.ru_maxrss, "wall": time.time() - start} - asm = (d / "m.s").read_text(errors="replace") - frames = [int(x, 16) if x.startswith("0x") else int(x) - for x in re.findall(r"sub[q]?\s+\$(0x[0-9a-f]+|\d+),\s*%rsp", asm)] - out["max_frame"] = max(frames) if frames else 0 - try: - run = subprocess.run([str(d / "m")], capture_output=True, text=True, timeout=30) - m = re.search(r"SIZE (\d+)", run.stdout) - if m: - out["run_size"] = int(m.group(1)) - except subprocess.TimeoutExpired: - pass - return out - - -def exponent(ns, values): - pts = [(math.log(n), math.log(v)) for n, v in zip(ns, values) if v and v > 0] - if len(pts) < 3: - return None - pts = pts[-3:] # the largest sizes: constant overheads dominate the small ones - mx = sum(x for x, _ in pts) / len(pts) - my = sum(y for _, y in pts) / len(pts) - den = sum((x - mx) ** 2 for x, _ in pts) - return sum((x - mx) * (y - my) for x, y in pts) / den if den else None - - -def one(job): - name, opt = job - gen, extra = GENERATORS[name] - sizes = [n for n in map(int, args.sizes.split(",")) if n <= CAP.get(name, 10**9)] - rows = [] - for n in sizes: - r = measure(gen(n), opt) - r["n"] = n - rows.append(r) - if r.get("timeout") or r.get("error"): - break - ok = [r for r in rows if "user" in r] - ns = [r["n"] for r in ok] - result = {"shape": name, "opt": opt, "rows": rows, - "k_time": exponent(ns, [max(r["user"] - ok[0]["user"] * 0.5, 1e-3) for r in ok]) if ok else None, - "k_rss": exponent(ns, [r["rss_kb"] for r in ok]), - "k_frame": exponent(ns, [r["max_frame"] for r in ok]) if extra == "frame" else None, - "k_size": exponent(ns, [r.get("run_size", 0) for r in ok]) if extra == "run-size" else None} - found = [] - if any(r.get("timeout") for r in rows): - found.append(f"timeout at N={rows[-1]['n']}") - if any(r.get("error") for r in rows): - found.append(f"error at N={rows[-1]['n']}: {rows[-1]['error'][-200:]}") - for key, limit in (("k_time", args.max_exponent), ("k_rss", args.max_exponent), ("k_frame", 1.3), ("k_size", 1.3)): - if result[key] is not None and result[key] > limit: - found.append(f"{key} = {result[key]:.2f}") - result["found"] = found - return result - - -def main(): - names = args.only.split(",") if args.only else list(GENERATORS) - jobs = [(n, int(o)) for n in names for o in args.opt.split(",")] - with ThreadPoolExecutor(args.jobs) as ex: - results = list(ex.map(one, jobs)) - (WORK / "results.json").write_text(json.dumps(results, indent=1)) - for r in results: - last = next((x for x in reversed(r["rows"]) if "user" in x), {}) - ks = " ".join(f"{k}={r[k]:.2f}" for k in ("k_time", "k_rss", "k_frame", "k_size") if r[k] is not None) - print(f"{r['shape']:15} O{r['opt']} N<={last.get('n')} user {last.get('user', 0):.1f}s " - f"rss {last.get('rss_kb', 0) // 1024}MB {ks} {'; '.join(r['found'])}") - - -main() diff --git a/rustc/solver-diff.py b/rustc/solver-diff.py deleted file mode 100644 index 0fec9da..0000000 --- a/rustc/solver-diff.py +++ /dev/null @@ -1,127 +0,0 @@ -#!/usr/bin/env python3 -"""Solver differential: the trait solvers and the borrow checkers must agree. - -Compiles each standalone UI test four ways: the old trait solver (`-Znext-solver=coherence`, -what compiletest pins), nightly's default (the new solver everywhere), the old solver with -Polonius (`-Zpolonius=next`), and the new solver with Polonius. Compared with the old solver -and NLL: - - ice a configuration crashes where the reference does not - verdict accepted by one and rejected by the other - codes both reject, with different sets of error codes (reported, not a finding: the - solvers word errors differently) - -A program accepted only by a non-reference configuration and runnable (it has `fn main`) is -interpreted with Miri under that configuration's flags: undefined behavior there means the -other configuration accepted something unsound (the Polonius soundness bugs had that shape). - -Tests that name a solver or Polonius in their headers (they test the difference on purpose) are -left out. - - rustc/solver-diff.py --rustc --tests /tests/ui --work [--only ] - [--known ] [--jobs 8] [--pause-on-finding] [--recheck] -""" - -import argparse -import json -import re -import shutil -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -CONFIGS = { - "old": ["-Znext-solver=coherence"], - "next": [], - "old-polonius": ["-Znext-solver=coherence", "-Zpolonius=next"], - "next-polonius": ["-Zpolonius=next"], -} -REFERENCE = "old" -KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def codes(stderr): - return sorted(set(re.findall(r"error\[(E\d{4})\]", stderr))) - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel, "kind": kind} - results = {} - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - for name, cfg in CONFIGS.items(): - # Metadata is enough for check tests (type checking and borrow checking run for it); - # build and run tests get a full build, which reaches monomorphization-time errors. - emit = "metadata" if kind in ("check-pass", "check-fail", None) else "link" - status, stderr, _ = uitest.compile(args.rustc, path.resolve(), d / name, flags, edition, cfg, - timeout=120, emit=emit) - results[name] = {"status": status, "codes": codes(stderr), "stderr": stderr[-2500:]} - record["status"] = {k: v["status"] for k, v in results.items()} - ref = results[REFERENCE] - found, notes = [], [] - for name, r in results.items(): - if name == REFERENCE: - continue - if r["status"] == "ice" and ref["status"] != "ice": - found.append({"config": name, "what": "ice", "stderr": r["stderr"]}) - elif r["status"] == "timeout" and ref["status"] != "timeout": - found.append({"config": name, "what": "timeout"}) - elif {r["status"], ref["status"]} == {"ok", "error"}: - f = {"config": name, "what": f"verdict: {REFERENCE} {ref['status']}, {name} {r['status']}", - "ref_codes": ref["codes"], "codes": r["codes"], - "stderr": (r if r["status"] == "error" else ref)["stderr"]} - accepted_by = name if r["status"] == "ok" else REFERENCE - if "fn main" in path.read_text(errors="replace"): - m = uitest.miri(path.resolve(), flags, edition, CONFIGS[accepted_by], timeout=120, cwd=WORK / "scratch") - f["miri"] = {"config": accepted_by, "status": m["status"], "stderr": m["stderr"][-1500:]} - if m["status"] == "ub": - f["what"] += f"; Miri: UB under {accepted_by}" - found.append(f) - elif r["status"] == ref["status"] == "error" and r["codes"] != ref["codes"]: - notes.append(f"{name} codes {r['codes']} vs {ref['codes']}") - record["found"] = [f"{f['config']}: {f['what']}" for f in found] - record["notes"] = notes - if found: - outdir = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(outdir, ignore_errors=True) - outdir.mkdir(parents=True) - shutil.copy(path, outdir / path.name) - (outdir / "finding.json").write_text(json.dumps( - {"test": rel, "flags": flags, "edition": edition, "configs": CONFIGS, "found": found}, indent=1)) - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - skip = lambda text, flags: re.search(r"next-solver|polonius|^//@\s*revisions:.*\bnext\b", text, re.M) - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, KINDS, skip): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests, configurations: {', '.join(CONFIGS)}", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/suggest-diff.py b/rustc/suggest-diff.py deleted file mode 100644 index 1112b61..0000000 --- a/rustc/suggest-diff.py +++ /dev/null @@ -1,195 +0,0 @@ -#!/usr/bin/env python3 -"""Suggestions apply: a machine-applicable suggestion must produce code that compiles the way the -suggestion promises. - -For each standalone UI test, collects the diagnostics rustc emits (`--error-format=json`) and, for -each one carrying a `MachineApplicable` suggestion inside the test file, applies that one -suggestion (all its parts) to a copy and compiles the copy again. Findings: - - lint-breaks the suggestion came from a warning (a lint) and the fixed file has an error the - original did not: a lint's fix must never break a build - parse the fixed file no longer parses - not-fixed the same diagnostic (code and message) is reported again at the edited place - ice the fixed file crashes the compiler - -Errors appearing after an *error's* suggestion is applied are expected (compilation gets further) -and are not reported. Tests with `//@ run-rustfix` are skipped by default: compiletest already -checks their fixes. - - rustc/suggest-diff.py --rustc --tests /tests/ui --work [--with-rustfix] - [--max 8] [--only ] [--known ] [--jobs 8] [--pause-on-finding] [--recheck] -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--with-rustfix", action="store_true") -p.add_argument("--max", type=int, default=8, help="suggestions tried per test") -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def diagnostics(source, flags, edition, out): - """(status, [diagnostic]) for `source`, metadata only.""" - out.mkdir(parents=True, exist_ok=True) - argv = [args.rustc, str(source), "--edition", edition or "2015", "--emit=metadata", "-o", str(out / "x"), - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", "--error-format=json", - *flags] - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=120, cwd=out, - env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) - except subprocess.TimeoutExpired: - return "timeout", [] - diags = [] - for line in r.stderr.splitlines(): - try: - diags.append(json.loads(line)) - except json.JSONDecodeError: - pass - if uitest.is_ice(r.stderr): - return "ice", diags - return ("ok" if r.returncode == 0 else "error"), diags - - -def suggestions(diag, file_name): - """The machine-applicable suggestions of a diagnostic, each a list of (start, end, text).""" - out = [] - for child in [diag] + diag.get("children", []): - parts = [(s["byte_start"], s["byte_end"], s["suggested_replacement"]) for s in child.get("spans", []) - if s.get("suggested_replacement") is not None - and s.get("suggestion_applicability") == "MachineApplicable" - and Path(s["file_name"]).name == file_name] - if parts: - out.append(sorted(parts)) - return out - - -def key(diag): - code = (diag.get("code") or {}).get("code") or "" - return code, diag["message"] - - -def primary(diag, file_name): - for s in diag.get("spans", []): - if s.get("is_primary") and Path(s["file_name"]).name == file_name: - return s["byte_start"], s["byte_end"] - return None - - -def errors(diags): - """The errors, by code (or lint name) when they have one: a renamed identifier changes the - message of the same lint.""" - return {(key(d)[0], "" if key(d)[0] else key(d)[1]) for d in diags - if d.get("level") == "error" and not d["message"].startswith("aborting")} - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel} - # The copy is compiled elsewhere: files named by relative path would be missing. - if re.search(r"^\s*(pub(\([^)]*\))?\s+)?mod\s+\w+\s*;|include(_str|_bytes)?!|#\[path", - path.read_text(errors="replace"), re.M): - record["skip"] = "uses files by path" - return record, [] - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - src = d / path.name - shutil.copy(path, src) - status, diags = diagnostics(src, flags, edition, d / "orig") - if status in ("ice", "timeout"): - record["skip"] = f"original {status}" - return record, [] - base_errors = errors(diags) - text = src.read_bytes() - found, tried = [], 0 - for diag in diags: - for parts in suggestions(diag, path.name): - if tried >= args.max: - break - tried += 1 - # Apply from the end, so earlier offsets stay valid; overlapping parts are skipped. - fixed, last = bytearray(text), None - ok = True - for start, end, repl in sorted(parts, reverse=True): - if last is not None and end > last: - ok = False - break - fixed[start:end] = repl.encode() - last = start - if not ok: - continue - fsrc = d / f"fix{tried}" / path.name - fsrc.parent.mkdir() - fsrc.write_bytes(bytes(fixed)) - fstatus, fdiags = diagnostics(fsrc, flags, edition, fsrc.parent) - what = None - if fstatus == "ice": - what = "ice" - elif any("expected" in m or "unexpected" in m or "unknown start of token" in m - for _, m in errors(fdiags) - base_errors): - what = "parse" - elif diag.get("level") == "warning" and errors(fdiags) - base_errors: - what = "lint-breaks" - elif sum(key(fd) == key(diag) for fd in fdiags) >= sum(key(od) == key(diag) for od in diags): - # Not one fewer of this diagnostic (nested braces legitimately report the next - # level, but there is then one fewer). - what = "not-fixed" - if what: - found.append({"what": what, "diagnostic": diag["message"], "code": key(diag)[0], - "level": diag.get("level"), "parts": parts, - "new_errors": sorted(c or m for c, m in errors(fdiags) - base_errors)[:5], - "fixed_name": f"fix{tried}.rs"}) - shutil.copy(fsrc, d / f"keep-fix{tried}.rs") - record["tried"] = tried - record["found"] = [f"{f['what']}: {f['code'] or f['level']} {f['diagnostic'][:80]}" for f in found] - if found: - out = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - shutil.copy(path, out / path.name) - for f in found: - shutil.copy(d / f"keep-{f['fixed_name']}", out / f["fixed_name"]) - (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, - "found": found}, indent=1)) - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - skip = None if args.with_rustfix else (lambda text, flags: re.search(r"^//@\s*run-rustfix", text, re.M)) - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, KINDS, skip): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/uitest.py b/rustc/uitest.py deleted file mode 100644 index 8ac1156..0000000 --- a/rustc/uitest.py +++ /dev/null @@ -1,161 +0,0 @@ -"""What the oracle scripts need from rustc's UI tests: their `//@` headers, which of them can be -compiled on their own on this host, and a way to build and run one. - -The header reading is the same as in ui-fuzz.py, ui-coverage.py and ui-solver-diff.py (the first -revision of a test with revisions). -""" - -import os -import re -import subprocess -from pathlib import Path - -# Tests that need more than one file, another target, or a tool this host may lack. -NOT_STANDALONE = re.compile( - r"^//@\s*(aux-build|aux-crate|aux-bin|aux-codegen-backend|proc-macro|add-minicore|" - r"needs-llvm-components|needs-sanitizer|needs-profiler|needs-rust-lld|needs-enzyme|" - r"only-(?!x86_64|linux|unix|64bit|elf|gnu)|ignore-x86_64|ignore-linux|ignore-unix|ignore-64bit|" - r"known-bug|rustc-env|unset-rustc-env)", re.M) - - -def headers(text): - """(flags, edition, kind, revision) of a test, for its first revision.""" - flags, edition, revision, kind = [], None, None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - elif key in ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail"): - kind = key - if revision: - flags += ["--cfg", revision] - return flags, edition, kind, revision - - -def tests(root, kinds, extra_skip=None): - """The standalone tests under `root` of the given kinds, as (path, flags, edition, kind).""" - root = Path(root) - for path in sorted(root.rglob("*.rs")): - if "auxiliary" in path.parts: - continue - text = path.read_text(errors="replace") - if NOT_STANDALONE.search(text): - continue - flags, edition, kind, _ = headers(text) - if kind not in kinds: - continue - if extra_skip and extra_skip(text, flags): - continue - yield path, flags, edition, kind - - -def is_ice(stderr): - return ("internal compiler error" in stderr or "the compiler unexpectedly panicked" in stderr - or "rustc interrupted by SIG" in stderr) - - -def compile(rustc, source, out, flags, edition, extra=(), timeout=300, emit="link"): - """Build `source` into the directory `out`; returns (status, stderr, binary). - status: ok, error, ice, timeout.""" - out.mkdir(parents=True, exist_ok=True) - binary = out / "prog" - argv = [str(rustc), str(source), "--edition", edition or "2015", f"--emit={emit}", "-o", str(binary), - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", "--error-format=short", - *flags, *extra] - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=timeout, cwd=out, - env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) - except subprocess.TimeoutExpired: - return "timeout", "", None - if is_ice(r.stderr): - return "ice", r.stderr, None - return ("ok" if r.returncode == 0 else "error"), r.stderr, binary if r.returncode == 0 else None - - -def run(binary, timeout=20): - """Run a built program; (exit, stdout, stderr), exit a number, or 'signal N' or 'timeout'.""" - try: - r = subprocess.run([str(binary)], capture_output=True, timeout=timeout, cwd=binary.parent, - stdin=subprocess.DEVNULL, env=dict(os.environ, RUST_BACKTRACE="0")) - except subprocess.TimeoutExpired: - return "timeout", b"", b"" - code = r.returncode if r.returncode >= 0 else f"signal {-r.returncode}" - return code, r.stdout, r.stderr - - -MIRI_TOOLCHAIN = Path.home() / ".rustup/toolchains/nightly-2026-10-06-x86_64-unknown-linux-gnu" -MIRI_SYSROOT = Path.home() / ".cache/miri" - - -def miri(source, flags, edition, extra=(), timeout=120, cwd=None): - """Interpret `source` with Miri (the pinned nightly's, sysroot from `cargo miri setup`). - Returns {status, exit, stdout, stderr}; status: ok (ran to the end, any exit code), ub, - unsupported, error (did not compile), timeout.""" - argv = [str(MIRI_TOOLCHAIN / "bin/miri"), "--sysroot", str(MIRI_SYSROOT), str(source), - "--edition", edition or "2015", "-Zunstable-options", "-Ainternal_features", - "-Aincomplete_features", "-Zmiri-disable-isolation", "-Zmiri-deterministic-floats", *flags, *extra] - try: - r = subprocess.run(argv, capture_output=True, timeout=timeout, cwd=cwd, stdin=subprocess.DEVNULL, - env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) - except subprocess.TimeoutExpired: - return {"status": "timeout", "exit": None, "stdout": "", "stderr": ""} - err = r.stderr.decode(errors="replace") - if "Undefined Behavior:" in err: - status = "ub" - elif "unsupported operation" in err or "can't call foreign function" in err: - status = "unsupported" - elif is_ice(err): - status = "ice" - elif re.search(r"^error(\[E\d+\])?: ", err, re.M) and r.returncode == 1 and "panicked" not in err: - status = "error" - else: - status = "ok" - return {"status": status, "exit": r.returncode, "stdout": r.stdout.decode(errors="replace"), "stderr": err} - - -def drive(todo, one, results, jobs, pause_on_finding): - """Run `one(*item)` for each item on `jobs` threads; each returns (record, findings). Records - go to the `results` file as JSON lines. With `pause_on_finding`, stops starting new items at - the first finding (the frontier loop) and returns 3; otherwise 0.""" - import json - import sys - from concurrent.futures import ThreadPoolExecutor, FIRST_COMPLETED, wait - it = iter(todo) - findings = done = 0 - stop = False - with ThreadPoolExecutor(jobs) as ex, open(results, "a") as out: - pending = set() - - def submit(): - nxt = None if stop else next(it, None) - if nxt is not None: - pending.add(ex.submit(one, *nxt)) - for _ in range(jobs * 2): - submit() - while pending: - finished, _ = wait(pending, return_when=FIRST_COMPLETED) - for fut in finished: - pending.discard(fut) - try: - record, found = fut.result() - except Exception as error: # a harness bug must not stop the run - record, found = {"test": "?", "harness": repr(error)[:300]}, [] - out.write(json.dumps(record) + "\n") - out.flush() - done += 1 - if found: - findings += 1 - print(f"FINDING {record.get('test')}: {record.get('found')}", flush=True) - stop = stop or pause_on_finding - if done % 100 == 0: - print(f"{done}/{len(todo)} done, {findings} with findings", flush=True) - submit() - print(f"{done} tests, {findings} with findings", flush=True) - return 3 if findings and pause_on_finding else 0 diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_52_34-1073611.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_52_34-1073611.txt new file mode 100644 index 0000000..a72e1e2 --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_52_34-1073611.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f74e2e5eb95 - ::create + 1: 0x7f74e2e5eae5 - ::force_capture + 2: 0x7f74e1b98fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f74e2e74fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f74e2e526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f74e2e4ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f74e2e5424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f74df8aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f74dfe60be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f74e4d15f4e - ::set_output_kind + 10: 0x7f74e4d32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f74e4a3f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f74e4a3240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f74e4732f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f74e4731eb2 - ::link + 15: 0x7f74e48ebec6 - ::link + 16: 0x7f74e48e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f74e491cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f74e491caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f74e4917d9b - ::new::thread_start + 20: 0x7f74dd889b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f74dd916ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_54_55-1317662.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_54_55-1317662.txt new file mode 100644 index 0000000..2305d9b --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_54_55-1317662.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7fd12bc5eb95 - ::create + 1: 0x7fd12bc5eae5 - ::force_capture + 2: 0x7fd12a998fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7fd12bc74fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7fd12bc526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7fd12bc4ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7fd12bc5424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7fd1286aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7fd128c60be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7fd12db15f4e - ::set_output_kind + 10: 0x7fd12db32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7fd12d83f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7fd12d83240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7fd12d532f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7fd12d531eb2 - ::link + 15: 0x7fd12d6ebec6 - ::link + 16: 0x7fd12d6e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7fd12d71cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7fd12d71caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7fd12d717d9b - ::new::thread_start + 20: 0x7fd126689b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7fd126716ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_57_29-1334708.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_57_29-1334708.txt new file mode 100644 index 0000000..f8961a3 --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_57_29-1334708.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7fdf2945eb95 - ::create + 1: 0x7fdf2945eae5 - ::force_capture + 2: 0x7fdf28198fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7fdf29474fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7fdf294526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7fdf2944ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7fdf2945424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7fdf25eaa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7fdf26460be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7fdf2b315f4e - ::set_output_kind + 10: 0x7fdf2b332a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7fdf2b03f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7fdf2b03240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7fdf2ad32f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7fdf2ad31eb2 - ::link + 15: 0x7fdf2aeebec6 - ::link + 16: 0x7fdf2aee5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7fdf2af1cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7fdf2af1caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7fdf2af17d9b - ::new::thread_start + 20: 0x7fdf23e89b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7fdf23f16ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_46-1343102.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_46-1343102.txt new file mode 100644 index 0000000..fe99e3f --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_46-1343102.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f0a5e05eb95 - ::create + 1: 0x7f0a5e05eae5 - ::force_capture + 2: 0x7f0a5cd98fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f0a5e074fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f0a5e0526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f0a5e04ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f0a5e05424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f0a5aaaa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f0a5b060be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f0a5ff15f4e - ::set_output_kind + 10: 0x7f0a5ff32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f0a5fc3f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f0a5fc3240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f0a5f932f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f0a5f931eb2 - ::link + 15: 0x7f0a5faebec6 - ::link + 16: 0x7f0a5fae5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f0a5fb1cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f0a5fb1caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f0a5fb17d9b - ::new::thread_start + 20: 0x7f0a58a89b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f0a58b16ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_59-1343572.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_59-1343572.txt new file mode 100644 index 0000000..3cef0de --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_59-1343572.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f3c0305eb95 - ::create + 1: 0x7f3c0305eae5 - ::force_capture + 2: 0x7f3c01d98fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f3c03074fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f3c030526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f3c0304ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f3c0305424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f3bffaaa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f3c00060be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f3c04f15f4e - ::set_output_kind + 10: 0x7f3c04f32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f3c04c3f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f3c04c3240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f3c04932f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f3c04931eb2 - ::link + 15: 0x7f3c04aebec6 - ::link + 16: 0x7f3c04ae5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f3c04b1cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f3c04b1caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f3c04b17d9b - ::new::thread_start + 20: 0x7f3bfda89b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f3bfdb16ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_16-1344795.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_16-1344795.txt new file mode 100644 index 0000000..ccbb9f7 --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_16-1344795.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7fb2b765eb95 - ::create + 1: 0x7fb2b765eae5 - ::force_capture + 2: 0x7fb2b6398fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7fb2b7674fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7fb2b76526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7fb2b764ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7fb2b765424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7fb2b40aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7fb2b4660be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7fb2b9515f4e - ::set_output_kind + 10: 0x7fb2b9532a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7fb2b923f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7fb2b923240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7fb2b8f32f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7fb2b8f31eb2 - ::link + 15: 0x7fb2b90ebec6 - ::link + 16: 0x7fb2b90e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7fb2b911cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7fb2b911caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7fb2b9117d9b - ::new::thread_start + 20: 0x7fb2b2089b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7fb2b2116ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_31-1345593.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_31-1345593.txt new file mode 100644 index 0000000..eb8596b --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_31-1345593.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7fb84445eb95 - ::create + 1: 0x7fb84445eae5 - ::force_capture + 2: 0x7fb843198fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7fb844474fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7fb8444526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7fb84444ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7fb84445424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7fb840eaa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7fb841460be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7fb846315f4e - ::set_output_kind + 10: 0x7fb846332a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7fb84603f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7fb84603240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7fb845d32f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7fb845d31eb2 - ::link + 15: 0x7fb845eebec6 - ::link + 16: 0x7fb845ee5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7fb845f1cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7fb845f1caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7fb845f17d9b - ::new::thread_start + 20: 0x7fb83ee89b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7fb83ef16ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T07_02_04-1382261.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T07_02_04-1382261.txt new file mode 100644 index 0000000..d206804 --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T07_02_04-1382261.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f6800c5eb95 - ::create + 1: 0x7f6800c5eae5 - ::force_capture + 2: 0x7f67ff998fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f6800c74fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f6800c526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f6800c4ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f6800c5424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f67fd6aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f67fdc60be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f6802b15f4e - ::set_output_kind + 10: 0x7f6802b32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f680283f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f680283240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f6802532f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f6802531eb2 - ::link + 15: 0x7f68026ebec6 - ::link + 16: 0x7f68026e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f680271cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f680271caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f6802717d9b - ::new::thread_start + 20: 0x7f67fb689b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f67fb716ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T07_10_33-1720154.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T07_10_33-1720154.txt new file mode 100644 index 0000000..bc9fd8d --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T07_10_33-1720154.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f60ede5eb95 - ::create + 1: 0x7f60ede5eae5 - ::force_capture + 2: 0x7f60ecb98fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f60ede74fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f60ede526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f60ede4ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f60ede5424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f60ea8aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f60eae60be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f60efd15f4e - ::set_output_kind + 10: 0x7f60efd32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f60efa3f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f60efa3240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f60ef732f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f60ef731eb2 - ::link + 15: 0x7f60ef8ebec6 - ::link + 16: 0x7f60ef8e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f60ef91cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f60ef91caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f60ef917d9b - ::new::thread_start + 20: 0x7f60e8889b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f60e8916ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink.py b/rustc/xlink.py deleted file mode 100644 index fab08ac..0000000 --- a/rustc/xlink.py +++ /dev/null @@ -1,123 +0,0 @@ -#!/usr/bin/env python3 -"""Cross-target build and link: every target rustc knows must build `core` and `alloc` and link a -program with its documented linker, with no undefined symbols. - -For each target, builds rustc/xlink-probe (a `no_std` program using 128-bit integers, float -conversions and math, float formatting, large copies and atomics, so that it needs the -compiler-builtins routines that go missing) with `cargo -Zbuild-std=core,alloc` and links it. -Targets whose default linker is a C compiler driver (absent for cross targets here) link with -`rust-lld` in the flavor their spec names instead. lld reports undefined symbols as errors, which -is what this looks for. - - link-undefined the link fails with undefined symbols - link the link fails otherwise - env a library or startup file of the target's C sysroot is missing here - build core or alloc (or the probe) does not compile for the target - ice the compiler crashes - - rustc/xlink.py --toolchain nightly-2026-10-06 --work [--targets t1,t2] [--jobs 6] - -Writes /results.json. Target directories are removed after each build (about 100 MB each). -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -from collections import Counter -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--toolchain", required=True) -p.add_argument("--work", required=True) -p.add_argument("--targets") -p.add_argument("--jobs", type=int, default=6) -args = p.parse_args() -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) -PROBE = Path(__file__).resolve().parent / "xlink-probe" -# Targets that need more than a target name (a CPU, an external linker that is not lld). -SKIP = re.compile(r"^(amdgcn|nvptx|bpf|spirv)|-uefi-|avr-none") - - -def rustc(*a): - return subprocess.run(["rustc", f"+{args.toolchain}", *a], capture_output=True, text=True, - env=dict(os.environ, RUSTC_BOOTSTRAP="1")) - - -def link_flags(spec): - """RUSTFLAGS for linking the probe with lld, from the target's spec.""" - flavor = spec.get("linker-flavor", "") - linker = spec.get("linker", "") - own_lld = "lld" in linker or flavor.endswith("-lld") or flavor in ("wasm-lld", "wasm-lld-cc") - flags = [] - if spec.get("is-like-wasm") or flavor.startswith("wasm"): - return flags + ["-Clink-arg=--no-entry", "-Clink-arg=--export=probe_entry"] - if flavor.startswith("msvc") or spec.get("is-like-msvc"): - if not own_lld: - flags += ["-Clinker=rust-lld", "-Clinker-flavor=lld-link"] - return flags + ["-Clink-arg=/ENTRY:probe_entry", "-Clink-arg=/NODEFAULTLIB"] - if flavor.startswith("darwin") or spec.get("is-like-darwin"): - if not own_lld: - flags += ["-Clinker=rust-lld", "-Clinker-flavor=ld64.lld"] - return flags + ["-Clink-arg=-e", "-Clink-arg=_probe_entry", "-Clink-arg=-undefined", - "-Clink-arg=dynamic_lookup"] - if not own_lld: - flags += ["-Clinker=rust-lld", "-Clinker-flavor=ld.lld"] - return flags + ["-Clink-arg=--entry=probe_entry"] - - -def one(target): - if SKIP.search(target): - return target, {"result": "skipped"} - out = rustc("--print", "target-spec-json", "-Zunstable-options", "--target", target) - if out.returncode != 0: - return target, {"result": "skipped", "why": "no spec"} - spec = json.loads(out.stdout) - flags = link_flags(spec) - tdir = WORK / "target" / target - env = dict(os.environ, CARGO_TARGET_DIR=str(tdir), RUSTFLAGS=" ".join(flags), CARGO_TERM_COLOR="never") - env.pop("RUSTC_WRAPPER", None) - try: - r = subprocess.run(["cargo", f"+{args.toolchain}", "build", "--release", "-Zbuild-std=core,alloc", - "-Zbuild-std-features=compiler-builtins-mem", "--target", target], - cwd=PROBE, env=env, capture_output=True, text=True, timeout=1500) - code, err = r.returncode, r.stderr - except subprocess.TimeoutExpired: - code, err = "timeout", "" - shutil.rmtree(tdir, ignore_errors=True) - if code == 0: - result = "ok" - elif "internal compiler error" in err or "panicked at" in err: - result = "ice" - elif "linking with" in err or "rust-lld: error" in err or "lld: error" in err: - if re.search(r"undefined (symbol|reference)", err): - result = "link-undefined" - elif re.search(r"unable to find library|cannot open crt|cannot open .*\.o\b|No such file", err): - # Libraries or startup files that ship with the target's prebuilt std or a C sysroot. - result = "env" - else: - result = "link" - else: - result = "build" - undefined = sorted(set(re.findall(r"undefined symbol: ([^\s\n]+)", err)))[:20] - first = next((l for l in err.splitlines() if re.search(r"error(\[|:)", l)), "")[:300] - return target, {"result": result, "flags": flags, "undefined": undefined, "first": first, - "tail": err[-2500:] if code != 0 else ""} - - -def main(): - targets = args.targets.split(",") if args.targets else rustc("--print", "target-list").stdout.split() - with ThreadPoolExecutor(args.jobs) as ex: - results = dict(ex.map(one, targets)) - (WORK / "results.json").write_text(json.dumps(results, indent=1)) - print(Counter(r["result"] for r in results.values())) - for t, r in sorted(results.items()): - if r["result"] not in ("ok", "skipped"): - print(f"{r['result']:15} {t:40} {' '.join(r['undefined'][:6]) or r['first'][:120]}") - - -main() From fe84b1847fd63a3eba825f9bf2c74f29dfd14d8a Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 07:49:52 +0000 Subject: [PATCH 12/23] mirth-lab callgraph: the reachability analysis in Rust (same counts and gap lists as callgraph.py on build-blk; ties ordered differently); coverage-report.sh uses it Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- crates/mirth-lab/src/main.rs | 4 + crates/mirth-lab/src/tools/callgraph.rs | 611 ++++++++++++++++++++++++ docs/coverage-handoff.md | 2 +- docs/coverage.md | 4 +- rustc/callgraph.py | 398 --------------- rustc/coverage-report.sh | 4 +- 6 files changed, 620 insertions(+), 403 deletions(-) create mode 100644 crates/mirth-lab/src/tools/callgraph.rs delete mode 100644 rustc/callgraph.py diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs index 6a82e55..c4dbcea 100644 --- a/crates/mirth-lab/src/main.rs +++ b/crates/mirth-lab/src/main.rs @@ -7,6 +7,7 @@ use clap::{Parser, Subcommand}; mod tools { pub mod abi_diff; + pub mod callgraph; pub mod crash_diff; pub mod diag_check; pub mod gate_check; @@ -59,6 +60,8 @@ enum Check { ScaleCheck(tools::scale_check::Args), /// rustc's extern "C" lowering matches clang's for random C signatures, per target. AbiDiff(tools::abi_diff::Args), + /// Reachability over the compiler's call graph; coverage of what can run, and gap lists. + Callgraph(tools::callgraph::Args), } fn main() -> ExitCode { @@ -78,6 +81,7 @@ fn main() -> ExitCode { Check::Xlink(a) => tools::xlink::run(a), Check::ScaleCheck(a) => tools::scale_check::run(a), Check::AbiDiff(a) => tools::abi_diff::run(a), + Check::Callgraph(a) => tools::callgraph::run(a), }; match result { Ok(code) => code, diff --git a/crates/mirth-lab/src/tools/callgraph.rs b/crates/mirth-lab/src/tools/callgraph.rs new file mode 100644 index 0000000..e93724a --- /dev/null +++ b/crates/mirth-lab/src/tools/callgraph.rs @@ -0,0 +1,611 @@ +//! Which of the compiler's functions can run at all: reachability over the call graph a compiler +//! built with rustc/callgraph.toml writes (`.graph`), against the functions (and blocks) +//! a compiler built with rustc/coverage.toml instruments. The functions that cannot be reached +//! leave coverage's denominator. +//! +//! The graph over-approximates what can run, so what it leaves out cannot run (as far as the +//! edges it knows go): direct calls, functions and closures used as values, callees MIR inlining +//! merged in, trait calls resolved in the caller; a call to a trait item reaches every body +//! implementing it, gated by rapid type analysis (a method of an impl for one of the compiler's +//! types only once reachable code builds the type, a trait impl's function only once reachable +//! code demands the trait for the impl's type). Roots: the compiler's and rustdoc's `main`s, +//! foreign-ABI functions, constant and static initializers, impls of traits from outside the +//! compiler, and what programs linking the compiler ran (`--external`). +//! +//! With hits, reports coverage of the reachable functions (and blocks), and checks the analysis: +//! a function that ran must be reachable. + +use std::collections::{BTreeMap, HashMap, HashSet, VecDeque}; +use std::fmt::Write as _; +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use regex::Regex; +use walkdir::WalkDir; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// The call-graph build's site directory (`.graph` files). + #[arg(long)] + graph: PathBuf, + /// The coverage build's site directory (`.sites` files). + #[arg(long)] + sites: PathBuf, + /// union.txt files of runs of the compiler. + #[arg(long)] + hit: Vec, + /// Directories of raw MIRTH_OUT logs. + #[arg(long)] + logs: Vec, + /// union.txt files of programs outside the compiler that link it: what they ran is a root. + #[arg(long)] + external: Vec, + #[arg(long)] + json: Option, + /// List the unreachable functions of these crates. + #[arg(long)] + unreachable: Vec, + /// Print how the graph reaches these functions. + #[arg(long)] + why: Vec, + /// Write the reachable functions that never ran, by crate and file. + #[arg(long)] + gaps: Option, + /// Write the blocks that never ran in functions that did. + #[arg(long)] + block_gaps: Option, +} + +const ROOTS: &[&str] = &["rustc_main::main", "rustc_driver_impl::main", "rustdoc::main"]; +/// Crates that do not run when the compiler does: proc macros, and a build-script helper. +const NOT_AT_RUN_TIME: &[&str] = + &["rustc_macros", "rustc_type_ir_macros", "rustc_index_macros", "rustc_windows_rc", "rustc_hir_macros", "rustc_fluent_macro"]; +const SCOPE: &[&str] = &["rustc_", "rustdoc"]; +/// Called by the language on any value, not through a bound: drop glue. +const UNGATED_TRAITS: &[&str] = &["core::ops::drop::Drop"]; + +static PARENT: LazyLock = LazyLock::new(|| Regex::new(r"^(.*)::\{[^}]*\}$").unwrap()); +static CRATE_ID: LazyLock = LazyLock::new(|| Regex::new(r"-[0-9a-f]{16}$").unwrap()); + +type Node = String; +type Cond = (u8, String, String); // (0: live type, 1: demand (trait, type)), as strings + +#[derive(Default)] +struct Graph { + path_of: HashMap, + hash_of: HashMap, + implements: HashMap, + external_impl: HashSet, + const_bodies: HashSet, + self_type: HashMap, + constructs: HashMap>, + spec_bounds: HashSet, + ending: HashMap, + diverges_into: HashMap>, + impl_key: HashMap, + demands: HashMap>, + edges: HashMap>, +} + +fn load_graph(dir: &PathBuf) -> Graph { + let mut g = Graph::default(); + let Ok(entries) = std::fs::read_dir(dir) else { return g }; + let mut files: Vec = entries.flatten().map(|e| e.path()).filter(|p| p.extension().is_some_and(|x| x == "graph")).collect(); + files.sort(); + for f in files { + let text = String::from_utf8_lossy(&std::fs::read(&f).unwrap_or_default()).into_owned(); + for line in text.lines() { + let p: Vec<&str> = line.split('\t').collect(); + match p[0] { + "body" if p.len() > 5 => { + let node = p[1].to_owned(); + if p.len() > 7 && SCOPE.iter().any(|s| p[7].starts_with(s)) { + g.self_type.insert(node.clone(), p[6].to_owned()); + } + // A trait with specializing impls: which impl a call reaches is decided by + // more than the bounds say, so such impls are not gated on them. + let specialized = p.len() > 11 && p[11] == "specialized"; + if p.len() > 10 && p[8] != "-" && p[10] != "-" && !UNGATED_TRAITS.contains(&p[9]) && !specialized { + g.impl_key.insert(node.clone(), (p[8].to_owned(), p[10].to_owned())); + } + g.path_of.insert(node.clone(), p[2].to_owned()); + g.hash_of.insert(p[2].to_owned(), node.clone()); + if p.len() > 12 && (p[12] == "ice-only" || p[12] == "diverges") { + g.ending.insert(node.clone(), p[12].to_owned()); + } + if p[5] == "const" || p[5] == "extern" { + g.const_bodies.insert(node.clone()); + } + if p[3] != "-" { + g.implements.insert(node.clone(), p[3].to_owned()); + if !p[4].starts_with("rustc_") { + g.external_impl.insert(node); + } + } + } + "diverges" if p.len() > 2 => { + g.diverges_into.entry(p[1].to_owned()).or_default().insert(p[2].to_owned()); + } + "specbound" if p.len() > 1 => { + g.spec_bounds.insert(p[1].to_owned()); + } + "demand" if p.len() > 3 => { + g.demands.entry(p[1].to_owned()).or_default().insert((p[2].to_owned(), p[3].to_owned())); + } + "edge" if p.len() > 3 => { + let map = if p[3] == "construct" { &mut g.constructs } else { &mut g.edges }; + map.entry(p[1].to_owned()).or_default().insert(p[2].to_owned()); + } + _ => {} + } + } + } + g +} + +fn parent(path: &str) -> Option { + PARENT.captures(path).map(|c| c[1].to_owned()) +} + +struct Site { + krate: String, + path: String, +} + +struct Block { + krate: String, + path: String, + name: String, + span: String, + snippet: String, + panics: bool, + logging: bool, +} + +fn load_sites(dir: &PathBuf) -> (BTreeMap, HashMap, BTreeMap) { + let (mut functions, mut span_of, mut blocks) = (BTreeMap::new(), HashMap::new(), BTreeMap::new()); + let Ok(entries) = std::fs::read_dir(dir) else { return (functions, span_of, blocks) }; + for e in entries.flatten() { + if e.path().extension().is_none_or(|x| x != "sites") { + continue; + } + let text = String::from_utf8_lossy(&std::fs::read(e.path()).unwrap_or_default()).into_owned(); + for line in text.lines() { + let f: Vec<&str> = line.split('\t').collect(); + if f.len() < 7 { + continue; + } + // Newer tables name the crate with its stable id (`rustc_hash-<16 hex digits>`). + let krate = CRATE_ID.replace(f[3], "").into_owned(); + match f[1] { + "cover" => { + functions.insert(f[0].to_owned(), Site { krate, path: f[4].to_owned() }); + span_of.insert(f[4].to_owned(), f[6].to_owned()); + } + "block" => { + let mut parts = f[5].split(' '); + let name = parts.next().unwrap_or("").to_owned(); + let tags: Vec<&str> = parts.collect(); + blocks.insert( + f[0].to_owned(), + Block { + krate, + path: f[4].to_owned(), + name, + span: f[6].to_owned(), + snippet: f.get(7).unwrap_or(&"").to_string(), + panics: tags.contains(&"panics"), + logging: tags.contains(&"log"), + }, + ); + } + _ => {} + } + } + } + (functions, span_of, blocks) +} + +fn words(path: &PathBuf) -> HashSet { + std::fs::read_to_string(path).unwrap_or_default().split_whitespace().map(str::to_owned).collect() +} + +pub fn run(args: Args) -> anyhow::Result { + let g = load_graph(&args.graph); + let bodies: HashSet<&Node> = g.path_of.keys().collect(); + // A body runs only on a compiler bug when every path ends in a panic, directly or through + // bodies that do. + let mut ice_nodes: HashSet = g.ending.iter().filter(|(_, e)| *e == "ice-only").map(|(n, _)| n.clone()).collect(); + loop { + let mut changed = false; + for n in g.ending.keys() { + if !ice_nodes.contains(n) + && let Some(into) = g.diverges_into.get(n) + && !into.is_empty() + && into.iter().all(|x| ice_nodes.contains(x)) + { + ice_nodes.insert(n.clone()); + changed = true; + } + } + if !changed { + break; + } + } + let ice_only: HashSet = ice_nodes.iter().filter_map(|n| g.path_of.get(n).cloned()).collect(); + let mut impl_key: HashMap = g.impl_key.iter().filter(|(_, k)| !g.spec_bounds.contains(&k.0)).map(|(n, k)| (n.clone(), k.clone())).collect(); + let mut implementors: HashMap<&Node, Vec<&Node>> = HashMap::new(); + for (body, item) in &g.implements { + implementors.entry(item).or_default().push(body); + } + let mut roots: HashSet = ROOTS.iter().filter_map(|r| g.hash_of.get(*r).cloned()).collect(); + roots.extend(g.external_impl.iter().cloned()); + roots.extend(g.const_bodies.iter().cloned()); + let mut children: HashMap> = HashMap::new(); + for (node, path) in &g.path_of { + let Some(mut up) = parent(path) else { continue }; + // Nested in a body, or in a static's or constant's initializer, which is not one here. + while !g.hash_of.contains_key(&up) { + match parent(&up) { + Some(p) => up = p, + None => break, + } + } + match g.hash_of.get(&up) { + Some(h) => { + children.entry(h.clone()).or_default().insert(node.clone()); + } + None => { + roots.insert(node.clone()); + } + } + } + let (functions, span_of, blocks) = load_sites(&args.sites); + let mut external: HashSet = HashSet::new(); + for u in &args.external { + external.extend(words(u)); + } + // Crates from crates.io that the scope takes in by name: other dependencies can name their + // types and need their impls without any bound in the compiler saying so. + let from_registry: HashSet<&str> = functions + .values() + .filter(|s| !span_of.get(&s.path).is_none_or(|sp| sp.starts_with("compiler/") || sp.starts_with("src/") || sp.starts_with("library/"))) + .map(|s| s.krate.as_str()) + .collect(); + impl_key.retain(|n, _| !g.path_of.get(n).is_some_and(|p| from_registry.contains(p.split("::").next().unwrap_or("")))); + let external_roots: HashSet = external.iter().filter_map(|x| functions.get(x)).filter_map(|s| g.hash_of.get(&s.path).cloned()).collect(); + roots.extend(external_roots.iter().cloned()); + + // Rapid type analysis. + let mut reachable: HashSet = HashSet::new(); + let mut live: HashSet = HashSet::new(); + let mut demanded: HashSet<(Node, Node)> = HashSet::new(); + let mut came_from: HashMap, &'static str)> = HashMap::new(); + let mut condition_from: HashMap = HashMap::new(); + let mut waiting: HashMap> = HashMap::new(); + let mut queue: VecDeque = VecDeque::new(); + let missing = |node: &Node, live: &HashSet, demanded: &HashSet<(Node, Node)>| -> Option { + if let Some(t) = g.self_type.get(node) + && !live.contains(t) + { + return Some((0, t.clone(), String::new())); + } + if let Some(k) = impl_key.get(node) + && !demanded.contains(k) + { + return Some((1, k.0.clone(), k.1.clone())); + } + None + }; + macro_rules! offer { + ($node:expr, $gated:expr, $source:expr, $how:expr) => {{ + let node: Node = $node; + if !reachable.contains(&node) { + came_from.entry(node.clone()).or_insert(($source, $how)); + match if $gated { missing(&node, &live, &demanded) } else { None } { + Some(c) => { + waiting.entry(c).or_default().insert(node); + } + None => queue.push_back(node), + } + } + }}; + } + let mut sorted_roots: Vec<&Node> = roots.iter().collect(); + sorted_roots.sort(); + for r in sorted_roots { + let gated = g.external_impl.contains(r) && !g.const_bodies.contains(r) && !external_roots.contains(r); + offer!(r.clone(), gated, None, "root"); + } + while let Some(node) = queue.pop_front() { + if !reachable.insert(node.clone()) { + continue; + } + let mut satisfied: Vec = Vec::new(); + if let Some(ts) = g.constructs.get(&node) { + for t in ts { + if live.insert(t.clone()) { + let c = (0, t.clone(), String::new()); + condition_from.insert(c.clone(), node.clone()); + satisfied.push(c); + } + } + } + if let Some(ks) = g.demands.get(&node) { + for k in ks { + if demanded.insert(k.clone()) { + let c = (1, k.0.clone(), k.1.clone()); + condition_from.insert(c.clone(), node.clone()); + satisfied.push(c); + } + } + } + for c in satisfied { + if let Some(nodes) = waiting.remove(&c) { + for n in nodes { + offer!(n, true, None, "root"); + } + } + } + let nexts: HashSet<&Node> = g.edges.get(&node).into_iter().flatten().chain(children.get(&node).into_iter().flatten()).collect(); + for nxt in nexts { + offer!(nxt.clone(), false, Some(node.clone()), "edge"); + // A call to a trait item reaches the bodies implementing it. + for imp in implementors.get(nxt).into_iter().flatten() { + offer!((*imp).clone(), true, Some(node.clone()), "dispatch"); + } + } + } + let reachable_paths: HashSet<&String> = reachable.iter().filter_map(|n| g.path_of.get(n)).collect(); + let mut out = String::new(); + for target in &args.why { + let mut node = g.hash_of.get(target).cloned(); + let ok = node.as_ref().is_some_and(|n| reachable.contains(n)); + let _ = writeln!(out, "\nwhy {target}:{}", if ok { "" } else { " not reachable" }); + let mut seen = HashSet::new(); + while let Some(n) = node.clone() { + if !reachable.contains(&n) || !seen.insert(n.clone()) { + break; + } + let (source, how) = came_from.get(&n).cloned().unwrap_or((None, "?")); + let mut extra = String::new(); + let conds = [g.self_type.get(&n).map(|t| (0u8, t.clone(), String::new())), impl_key.get(&n).map(|k| (1u8, k.0.clone(), k.1.clone()))]; + for c in conds.into_iter().flatten() { + if let Some(from) = condition_from.get(&c) { + let _ = write!(extra, " [{} from {}]", if c.0 == 0 { "live" } else { "demand" }, g.path_of.get(from).unwrap_or(from)); + } + } + let _ = writeln!(out, " {} <- {how}{extra}", g.path_of.get(&n).unwrap_or(&n)); + node = source; + } + } + let functions: BTreeMap<&String, &Site> = functions.iter().filter(|(_, s)| !NOT_AT_RUN_TIME.contains(&s.krate.as_str())).collect(); + let paths: HashSet<&String> = functions.values().map(|s| &s.path).collect(); + let known: HashSet<&String> = paths.iter().filter(|p| g.hash_of.contains_key(**p)).copied().collect(); + let unreach: HashSet<&String> = known.iter().filter(|p| !reachable_paths.contains(**p)).copied().collect(); + let _ = writeln!(out, "{} bodies in the graph, {} roots, {} reachable", bodies.len(), roots.len(), reachable.iter().filter(|n| bodies.contains(n)).count()); + let _ = writeln!( + out, + "{} instrumented functions; {} in the graph; {} unreachable ({:.1}%)", + functions.len(), + known.len(), + unreach.len(), + 100.0 * unreach.len() as f64 / known.len().max(1) as f64 + ); + let mut hit: HashSet = external.clone(); + for u in &args.hit { + hit.extend(words(u)); + } + for d in &args.logs { + for e in WalkDir::new(d).into_iter().filter_map(Result::ok) { + if e.path().extension().is_some_and(|x| x == "log") { + for line in String::from_utf8_lossy(&std::fs::read(e.path()).unwrap_or_default()).lines() { + if let Some(s) = line.strip_prefix("V\t") { + hit.insert(s.to_owned()); + } + } + } + } + } + let hit_paths: HashSet<&String> = hit.iter().filter_map(|s| functions.get(s)).map(|s| &s.path).collect(); + let mut by_crate: BTreeMap<&str, (usize, usize)> = BTreeMap::new(); + for s in functions.values() { + let row = by_crate.entry(s.krate.as_str()).or_default(); + if unreach.contains(&s.path) { + continue; + } + row.0 += 1; + row.1 += hit_paths.contains(&s.path) as usize; + } + if !hit.is_empty() { + let total: usize = by_crate.values().map(|r| r.0).sum(); + let ran: usize = by_crate.values().map(|r| r.1).sum(); + let _ = writeln!(out, "coverage: {} functions ran; of the {total} reachable ones, {ran} ({:.1}%)", hit_paths.len(), 100.0 * ran as f64 / total.max(1) as f64); + let mut wrong: Vec<&&String> = hit_paths.iter().filter(|p| unreach.contains(**p)).collect(); + wrong.sort(); + let ice: HashSet<&String> = functions.values().map(|s| &s.path).filter(|p| ice_only.contains(*p) && !unreach.contains(p)).collect(); + let ice_ran = ice.iter().filter(|p| hit_paths.contains(**p)).count(); + let _ = writeln!( + out, + "of the reachable ones, {} only panic (they run on a compiler bug; {ice_ran} ran): without them, {} of {} ({:.1}%)", + ice.len(), + ran - ice_ran, + total - ice.len(), + 100.0 * (ran - ice_ran) as f64 / (total - ice.len()).max(1) as f64 + ); + let _ = writeln!(out, "ran although unreachable (edges the analysis misses): {}", wrong.len()); + for w in wrong.iter().take(30) { + let _ = writeln!(out, " {w}"); + } + if !blocks.is_empty() { + // Blocks: each function's entry (its own site) and its other blocks, in reachable + // functions; a block only panics when every path from it does, or its function does. + let mut rows: BTreeMap<&str, [usize; 4]> = BTreeMap::new(); + for s in functions.values() { + if !unreach.contains(&s.path) { + let r = rows.entry(s.krate.as_str()).or_default(); + let (h, i) = (hit_paths.contains(&s.path), ice_only.contains(&s.path)); + r[0] += 1; + r[1] += h as usize; + r[2] += i as usize; + r[3] += (i && h) as usize; + } + } + let (mut logged, mut logged_ran) = (0, 0); + for (site, b) in &blocks { + if NOT_AT_RUN_TIME.contains(&b.krate.as_str()) || unreach.contains(&b.path) || !paths.contains(&b.path) { + continue; + } + if b.logging { + logged += 1; + logged_ran += hit.contains(site) as usize; + continue; + } + let r = rows.entry(b.krate.as_str()).or_default(); + let (h, p) = (hit.contains(site), b.panics || ice_only.contains(&b.path)); + r[0] += 1; + r[1] += h as usize; + r[2] += p as usize; + r[3] += (p && h) as usize; + } + let s: [usize; 4] = rows.values().fold([0; 4], |a, r| [a[0] + r[0], a[1] + r[1], a[2] + r[2], a[3] + r[3]]); + let _ = writeln!( + out, + "blocks: of the {} in reachable functions, {} ran ({:.1}%); {} only panic ({} ran): without them, {} of {} ({:.1}%); not counted: {logged} blocks of logging macros, which run only with RUSTC_LOG ({logged_ran} ran)", + s[0], s[1], 100.0 * s[1] as f64 / s[0].max(1) as f64, s[2], s[3], s[1] - s[3], s[0] - s[2], + 100.0 * (s[1] - s[3]) as f64 / (s[0] - s[2]).max(1) as f64 + ); + let _ = writeln!(out, "{:40} {:>7} {:>9} {:>6} (panic-only blocks aside)", "crate", "ran", "blocks", "%"); + let mut sorted: Vec<(&&str, &[usize; 4])> = rows.iter().collect(); + sorted.sort_by(|a, b| { + let f = |r: &[usize; 4]| (r[1] - r[3]) as f64 / (r[0] - r[2]).max(1) as f64; + f(a.1).partial_cmp(&f(b.1)).unwrap() + }); + for (k, r) in sorted { + if r[0] > r[2] { + let _ = writeln!(out, "{k:40} {:7} {:9} {:6.1}", r[1] - r[3], r[0] - r[2], 100.0 * (r[1] - r[3]) as f64 / (r[0] - r[2]) as f64); + } + } + } + let _ = writeln!(out, "{:40} {:>7} {:>9} {:>6}", "crate", "ran", "reachable", "%"); + let mut sorted: Vec<(&&str, &(usize, usize))> = by_crate.iter().collect(); + sorted.sort_by(|a, b| (a.1.1 as f64 / a.1.0.max(1) as f64).partial_cmp(&(b.1.1 as f64 / b.1.0.max(1) as f64)).unwrap()); + for (k, (n, r)) in sorted { + if *n > 0 { + let _ = writeln!(out, "{k:40} {r:7} {n:9} {:6.1}", 100.0 * *r as f64 / *n as f64); + } + } + } + for krate in &args.unreachable { + let _ = writeln!(out, "\nunreachable in {krate}:"); + let mut list: Vec<&&String> = unreach.iter().filter(|p| p.starts_with(&format!("{krate}::"))).collect(); + list.sort(); + for p in list { + let _ = writeln!(out, " {p}"); + } + } + if let Some(j) = &args.json { + let sorted = |set: Vec<&String>| -> Vec { + let mut v: Vec = set.into_iter().cloned().collect(); + v.sort(); + v + }; + let json = serde_json::json!({ + "unreachable": sorted(unreach.iter().copied().collect()), + "ice_only": sorted(ice_only.iter().filter(|p| paths.contains(p)).collect()), + "ran_unreachable": sorted(hit_paths.iter().filter(|p| unreach.contains(**p)).copied().collect()), + "reachable_not_hit": sorted(paths.iter().filter(|p| !unreach.contains(**p) && !hit_paths.contains(**p)).copied().collect()), + }); + std::fs::write(j, serde_json::to_string(&json)?)?; + } + if let Some(gp) = &args.gaps { + let mut files: BTreeMap<(&str, String), Vec<(String, &String)>> = BTreeMap::new(); + for s in functions.values() { + if !unreach.contains(&s.path) && !hit_paths.contains(&s.path) { + let span = span_of.get(&s.path).cloned().unwrap_or_else(|| "?".into()); + let file = span.rsplitn(3, ':').last().unwrap_or("").to_owned(); + files.entry((s.krate.as_str(), file)).or_default().push((span, &s.path)); + } + } + std::fs::write(gp, gaps_text(&files, |span, path| { + let tag = if ice_only.contains(*path) { " (only panics)" } else { "" }; + let line = span.rsplitn(3, ':').nth(1).unwrap_or(""); + format!("- `{path}` {line}{tag}") + }, "Reachable functions that never ran"))?; + let _ = writeln!(out, "gaps written to {}", gp.display()); + } + if let Some(bp) = &args.block_gaps + && !blocks.is_empty() + { + let mut files: BTreeMap<(&str, String), Vec<(String, String)>> = BTreeMap::new(); + for (site, b) in &blocks { + if hit_paths.contains(&b.path) && !hit.contains(site) && !NOT_AT_RUN_TIME.contains(&b.krate.as_str()) && !b.logging { + let file = b.span.rsplitn(3, ':').last().unwrap_or("").to_owned(); + let panics = b.panics || ice_only.contains(&b.path); + let snippet: String = b.snippet.chars().take(100).collect(); + files.entry((b.krate.as_str(), file)).or_default().push(( + b.span.clone(), + format!("`{}` {}{}: `{snippet}`", b.path, b.name, if panics { " (only panics)" } else { "" }), + )); + } + } + // The lines with the most first within a file. + let line_no = |span: &str| -> (u64, u64) { + let p: Vec<&str> = span.rsplitn(3, ':').collect(); + (p.get(1).and_then(|x| x.parse().ok()).unwrap_or(0), p.first().and_then(|x| x.parse().ok()).unwrap_or(0)) + }; + let mut text = format!("# Blocks that never ran in functions that did: {}\n\n", files.values().map(Vec::len).sum::()); + let mut by_crate: BTreeMap<&str, usize> = BTreeMap::new(); + for ((k, _), v) in &files { + *by_crate.entry(k).or_default() += v.len(); + } + let mut crates: Vec<(&&str, &usize)> = by_crate.iter().collect(); + crates.sort_by(|a, b| b.1.cmp(a.1)); + for (krate, n) in crates { + let _ = writeln!(text, "## {krate} ({n})\n"); + let mut fs: Vec<(&(&str, String), &Vec<(String, String)>)> = files.iter().filter(|((k, _), _)| k == krate).collect(); + fs.sort_by(|a, b| b.1.len().cmp(&a.1.len())); + for ((_, file), entries) in fs { + let _ = writeln!(text, "### {file} ({})\n", entries.len()); + let mut es = entries.clone(); + es.sort_by_key(|(span, _)| line_no(span)); + for (span, desc) in es { + let _ = writeln!(text, "- {} {desc}", line_no(&span).0); + } + text.push('\n'); + } + } + std::fs::write(bp, text)?; + let _ = writeln!(out, "block gaps written to {}", bp.display()); + } + print!("{out}"); + Ok(ExitCode::SUCCESS) +} + +/// A gap list: by crate (largest first), then file (largest first), entries sorted by span. +fn gaps_text

(files: &BTreeMap<(&str, String), Vec<(String, P)>>, entry: impl Fn(&str, &P) -> String, title: &str) -> String { + let total: usize = files.values().map(Vec::len).sum(); + let mut text = format!("# {title}: {total}\n\n"); + let mut by_crate: BTreeMap<&str, usize> = BTreeMap::new(); + for ((k, _), v) in files { + *by_crate.entry(k).or_default() += v.len(); + } + let mut crates: Vec<(&&str, &usize)> = by_crate.iter().collect(); + crates.sort_by(|a, b| b.1.cmp(a.1)); + for (krate, n) in crates { + let _ = writeln!(text, "## {krate} ({n})\n"); + let mut fs: Vec<(&(&str, String), &Vec<(String, P)>)> = files.iter().filter(|((k, _), _)| k == krate).collect(); + fs.sort_by(|a, b| b.1.len().cmp(&a.1.len())); + for ((_, file), entries) in fs { + let _ = writeln!(text, "### {file} ({})\n", entries.len()); + let mut es: Vec<&(String, P)> = entries.iter().collect(); + es.sort_by(|a, b| a.0.cmp(&b.0)); + for (span, p) in es { + let _ = writeln!(text, "{}", entry(span, p)); + } + text.push('\n'); + } + } + text +} diff --git a/docs/coverage-handoff.md b/docs/coverage-handoff.md index 6fe60af..f68b6fe 100644 --- a/docs/coverage-handoff.md +++ b/docs/coverage-handoff.md @@ -158,7 +158,7 @@ MIRTH_RUNTIME=$WORK/build-cg/mirth-runtime/libmirth_runtime.rlib MIRTH_WATCH=cg. Measure each with `coverage-report.sh` and keep the ones that add. 2. **Tighten the denominator** where the graph says reachable but nothing can run it. For each suspicious class, `--why` a sample: if the chain has a spurious step (a demand nothing really - makes), fix the analysis in `sites.rs` (`graph`) or `callgraph.py`, rebuild the graph, and + makes), fix the analysis in `sites.rs` (`graph`) or `mirth-lab callgraph`, rebuild the graph, and check 0 misses. Classes to look at first: the AST's encode/decode, `handle_cycle_error` per query, `format_value` (only on a verification failure or `RUSTC_VERIFY_REUSE`), derived impls of types only built in tests. A function that only runs on a compiler bug is already diff --git a/docs/coverage.md b/docs/coverage.md index 76db752..c6c9ee1 100644 --- a/docs/coverage.md +++ b/docs/coverage.md @@ -123,7 +123,7 @@ of rustc-hash share a name): `T: From`, a derive's `T: Encodable` for every field) and of the trait's associated types (`type Domain: JoinSemiLattice`), recursively. -`rustc/callgraph.py` computes reachability from the compiler's `main`s, every function with a +`mirth-lab callgraph` computes reachability from the compiler's `main`s, every function with a foreign ABI (callbacks from C, C++ and LLVM), every implementation of a trait from outside the compiler (which `std` may call), and every initializer. A call to a trait item reaches its implementations (class hierarchy analysis), but an implementation for one of the compiler's @@ -207,5 +207,5 @@ logs the same way. **All together, with sink and its option configurations: 50,762 of the 62,516 functions that can run (81.2%); without the 911 that only panic, 50,715 of 61,605 (82.3%).** -`rustc/coverage-report.sh` recomputes this; `rustc/callgraph.py --gaps ` lists the rest +`rustc/coverage-report.sh` recomputes this; `mirth-lab callgraph --gaps ` lists the rest by crate and file, largest first. What is left, and the plan for it: [coverage-handoff.md](coverage-handoff.md). diff --git a/rustc/callgraph.py b/rustc/callgraph.py deleted file mode 100644 index b658ed6..0000000 --- a/rustc/callgraph.py +++ /dev/null @@ -1,398 +0,0 @@ -#!/usr/bin/env python3 -"""Which of the compiler's functions can run at all: reachability over the call graph that a -compiler built with rustc/callgraph.toml writes (`.graph`), against the functions a -compiler built with rustc/coverage.toml instruments (`cover` sites). The functions that cannot -be reached are taken out of coverage's denominator. - - rustc/callgraph.py --graph /mirth-sites --sites /mirth-sites - [--hit ...] [--external ...] [--logs ...] - [--json out.json] [--unreachable ] [--why ] [--gaps ] - -The graph over-approximates what can run, so what it leaves out cannot run (as far as the -edges it knows go): - -- edges: direct calls, functions and closures used as values, callees MIR inlining merged in, - and trait calls resolved in the caller's context; -- a call to a trait item reaches every body implementing it, and the trait's own default body; - but (rapid type analysis) a method of an impl for one of the compiler's structs or enums only - once reachable code builds that type (an aggregate, a constructor, a constant of it), and a - function of a trait impl only once reachable code demands the trait for the impl's type (a - call whose bounds say so, an impl selected for one, a cast to `dyn Trait`): see - docs/coverage.md for the exceptions; -- a body nested in another (a closure, an inline const) is reached with it, and one nested in - something that is not a body (a static's or a constant's initializer) is a root; -- roots: the compiler's and rustdoc's `main`s, every function with a foreign ABI (callbacks from C, C++ and - LLVM), every body implementing a trait from outside the compiler (for one of the compiler's - types: once the type is built), - which the standard library may call (`Iterator::next`, `Drop::drop`, `Debug::fmt`, ...), and - every constant's and static's initializer (tables of function pointers, callbacks). - -With --hit or --logs, reports coverage of the reachable functions, and checks the analysis: a -function that ran must be reachable; any that are not are listed (an edge kind it misses). -""" - -import argparse -import json -import re -from collections import defaultdict, deque -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--graph", required=True) -p.add_argument("--sites", required=True) -p.add_argument("--hit", action="append", default=[]) -p.add_argument("--logs", action="append", default=[]) -p.add_argument("--external", action="append", default=[], - help="union.txt of programs outside the compiler that link it (ui-fulldeps): what they " - "ran counts, and is a root, since their own mains call it") -p.add_argument("--json") -p.add_argument("--unreachable", action="append", default=[]) -p.add_argument("--why", action="append", default=[], help="print how a function is reached") -p.add_argument("--gaps", help="write the reachable functions that never ran, by crate and file, to this file") -p.add_argument("--block-gaps", help="write the blocks that never ran in functions that did, by crate and file, " - "to this file (a compiler built with `[coverage] blocks`)") -args = p.parse_args() - -ROOTS = {"rustc_main::main", "rustc_driver_impl::main", "rustdoc::main"} -# Crates that do not run when the compiler does: proc macros (run while it is built) and the -# Windows resource helper of its build script. -NOT_AT_RUN_TIME = {"rustc_macros", "rustc_type_ir_macros", "rustc_index_macros", "rustc_windows_rc", - "rustc_hir_macros", "rustc_fluent_macro"} - -# Nodes are DefPathHashes; bodies also have their path (as coverage sites name them). -path_of, hash_of = {}, {} -implements, external_impl, const_bodies = {}, set(), set() -SCOPE = ("rustc_", "rustdoc") -self_type = {} # a method in an impl for one of the compiler's structs or enums -> that type -constructs = defaultdict(set) # body -> types it builds -spec_bounds = set() # traits a specializing impl's bounds name -impl_trait = {} -ending = {} # a body no path of which returns -> "ice-only" (it panics) or "diverges" -diverges_into = defaultdict(set) # such a body -> the other functions ending its paths -impl_key = {} # a function of a trait impl -> (trait, the struct or enum the impl is for) -demands = defaultdict(set) # body -> (trait, type) pairs it needs implemented -# Called by the language on any value, not through a bound: drop glue. -UNGATED_TRAITS = {"core::ops::drop::Drop"} -edges = defaultdict(set) -for f in Path(args.graph).glob("*.graph"): - for line in f.read_text(errors="replace").splitlines(): - parts = line.split("\t") - if parts[0] == "body": - node, path, item, item_path, kind = parts[1], parts[2], parts[3], parts[4], parts[5] - if len(parts) > 7 and parts[7].startswith(SCOPE): - self_type[node] = parts[6] - # A trait with specializing impls: which impl a call reaches is decided by more than - # the bounds say, so such impls are not gated on them. - specialized = len(parts) > 11 and parts[11] == "specialized" - if (len(parts) > 10 and parts[8] != "-" and parts[10] != "-" and parts[9] not in UNGATED_TRAITS - and not specialized): - impl_key[node] = (parts[8], parts[10]) - - path_of[node] = path - hash_of[path] = node - if len(parts) > 12 and parts[12] in ("ice-only", "diverges"): - ending[node] = parts[12] - if kind in ("const", "extern"): - const_bodies.add(node) - if item != "-": - implements[node] = item - if not item_path.startswith("rustc_"): - external_impl.add(node) - elif parts[0] == "diverges": - diverges_into[parts[1]].add(parts[2]) - elif parts[0] == "specbound": - spec_bounds.add(parts[1]) - elif parts[0] == "demand": - demands[parts[1]].add((parts[2], parts[3])) - elif parts[0] == "edge": - if parts[3] == "construct": - constructs[parts[1]].add(parts[2]) - else: - edges[parts[1]].add(parts[2]) -bodies = set(path_of) -# A body runs only on a compiler bug when every path ends in a panic, directly or through bodies -# that do (`default_extern_query` and the closures that call it). -ice_nodes = {n for n, e in ending.items() if e == "ice-only"} -changed = True -while changed: - changed = False - for node, e in ending.items(): - if node not in ice_nodes and diverges_into[node] and diverges_into[node] <= ice_nodes: - ice_nodes.add(node) - changed = True -ice_only = {path_of[n] for n in ice_nodes} -impl_key = {node: key for node, key in impl_key.items() if key[0] not in spec_bounds} - -implementors = defaultdict(set) -for body, item in implements.items(): - implementors[item].add(body) - - -def parent(path): - m = re.match(r"^(.*)::\{[^}]*\}$", path) - return m.group(1) if m else None - - -children = defaultdict(set) -roots = {hash_of[r] for r in ROOTS if r in hash_of} | external_impl | const_bodies -for node, path in path_of.items(): - up = parent(path) - if up is None: - continue - # Nested in a body, or in a static's or a constant's initializer, which is not one here. - while up is not None and up not in hash_of and parent(up) is not None: - up = parent(up) - if up in hash_of: - children[hash_of[up]].add(node) - else: - roots.add(node) - -functions = {} # site -> (crate, path) -span_of = {} -blocks = {} # site -> (crate, function path, block, span, snippet, only panics) -logging = set() # blocks all of whose code is a logging macro's: they run only with RUSTC_LOG -for table in Path(args.sites).glob("*.sites"): - for line in table.read_text(errors="replace").splitlines(): - f = line.split("\t") - if len(f) < 7: - continue - # Newer tables name the crate with its stable id (`rustc_hash-<16 hex digits>`). - krate = re.sub(r"-[0-9a-f]{16}$", "", f[3]) - if f[1] == "cover": - functions[f[0]] = (krate, f[4]) - span_of[f[4]] = f[6] - elif f[1] == "block": - name, *tags = f[5].split(" ") - if "log" in tags: - logging.add(f[0]) - blocks[f[0]] = (krate, f[4], name, f[6], f[7] if len(f) > 7 else "", "panics" in tags) -external = set() -for u in args.external: - external |= set(Path(u).read_text().split()) - -# Crates from crates.io that the scope takes in by name (rustc-hash, rustc-stable-hash): other -# dependencies, which the graph does not cover, can name their types and need their impls -# without any bound in the compiler saying so. Their impls are not gated on demand. -from_registry = {krate for _, (krate, path) in functions.items() - if not span_of.get(path, "compiler/").startswith(("compiler/", "src/", "library/"))} -impl_key = {node: key for node, key in impl_key.items() - if path_of[node].split("::", 1)[0] not in from_registry} - -# What programs outside the compiler ran of it: their mains call it, so it is a root. -external_roots = {hash_of[functions[x][1]] for x in external if x in functions and functions[x][1] in hash_of} -roots |= external_roots - -# Rapid type analysis: a method of an impl for one of the compiler's types counts for trait -# dispatch (and as an external trait's root) only once reachable code builds the type; and a -# function of a trait impl only once reachable code needs that type to implement that trait (a -# call whose bounds say so, a cast to `dyn Trait`, the trait's item used with that `Self`). A -# call the caller's types resolve to the implementation reaches it directly. -reachable, live, demanded = set(), set(), set() -came_from = {} # node -> (from node, how) -condition_from = {} # condition -> the body that met it -waiting = defaultdict(set) # condition -> functions waiting for it -queue = deque() - - -def missing(node): - out = [] - t = self_type.get(node) - if t is not None and t not in live: - out.append(("live", t)) - k = impl_key.get(node) - if k is not None and k not in demanded: - out.append(("demand", k)) - return out - - -def offer(node, gated, source=None, how="root"): - if node in reachable: - return - came_from.setdefault(node, (source, how)) - lacking = missing(node) if gated else [] - if lacking: - waiting[lacking[0]].add(node) - else: - queue.append(node) - - -def satisfied(condition): - for node in waiting.pop(condition, ()): - offer(node, True) - - -for r in roots: - offer(r, r in external_impl and r not in const_bodies and r not in external_roots) -while queue: - node = queue.popleft() - if node in reachable: - continue - reachable.add(node) - for t in constructs.get(node, ()): - if t not in live: - live.add(t) - condition_from[("live", t)] = node - satisfied(("live", t)) - for k in demands.get(node, ()): - if k not in demanded: - demanded.add(k) - condition_from[("demand", k)] = node - satisfied(("demand", k)) - for nxt in edges.get(node, set()) | children.get(node, set()): - offer(nxt, False, node, "edge") - # A call to a trait item reaches the bodies implementing it. - for impl in implementors.get(nxt, ()): - offer(impl, True, node, "dispatch") -reachable_paths = {path_of[n] for n in reachable if n in path_of} -for target in args.why: - node = hash_of.get(target) - print(f"\nwhy {target}:" + ("" if node in reachable else " not reachable")) - seen = set() - while node is not None and node in reachable and node not in seen: - seen.add(node) - source, how = came_from.get(node, (None, "?")) - extra = "" - if True: - for c in [("live", self_type.get(node)), ("demand", impl_key.get(node))]: - if c[1] is not None and c in condition_from: - extra += f" [{c[0]} from {path_of.get(condition_from[c], condition_from[c])}]" - print(f" {path_of.get(node, node)} <- {how}{extra}") - node = source - -functions = {s: v for s, v in functions.items() if v[0] not in NOT_AT_RUN_TIME} -paths = {path for _, path in functions.values()} -known = paths & set(hash_of) -unreach = {path for path in known if path not in reachable_paths} -print(f"{len(bodies)} bodies in the graph, {len(roots)} roots, {len(reachable & bodies)} reachable") -print(f"{len(functions)} instrumented functions; {len(known)} in the graph; " - f"{len(unreach)} unreachable ({100 * len(unreach) / max(len(known), 1):.1f}%)") - -hit = set(external) -for u in args.hit: - hit |= set(Path(u).read_text().split()) -for d in args.logs: - for log in Path(d).rglob("*.log"): - for line in log.read_text(errors="replace").splitlines(): - if line.startswith("V\t"): - hit.add(line[2:]) -hit_paths = {functions[s][1] for s in hit if s in functions} -by_crate = defaultdict(lambda: [0, 0, 0]) -for site, (krate, path) in functions.items(): - row = by_crate[krate] - if path in unreach: - continue - row[0] += 1 - row[1] += path in hit_paths -if hit: - total = sum(r[0] for r in by_crate.values()) - ran = sum(r[1] for r in by_crate.values()) - print(f"coverage: {len(hit_paths)} functions ran; of the {total} reachable ones, {ran} " - f"({100 * ran / max(total, 1):.1f}%)") - wrong = sorted(hit_paths & unreach) - ice = {path for _, (k, path) in functions.items() if path in ice_only and path not in unreach} - ice_ran = ice & hit_paths - print(f"of the reachable ones, {len(ice)} only panic (they run on a compiler bug; {len(ice_ran)} ran): " - f"without them, {ran - len(ice_ran)} of {total - len(ice)} " - f"({100 * (ran - len(ice_ran)) / max(total - len(ice), 1):.1f}%)") - print(f"ran although unreachable (edges the analysis misses): {len(wrong)}") - for w in wrong[:30]: - print(" ", w) - if blocks: - # Blocks: each function's entry (its own site) and its other blocks, in the functions - # that can run. A block only panics when every path from it does, or its function does. - rows = defaultdict(lambda: [0, 0, 0, 0]) # crate -> reachable, ran, panic-only, panic-only ran - for site, (krate, path) in functions.items(): - if path not in unreach: - row = rows[krate] - row[0] += 1 - row[1] += path in hit_paths - row[2] += path in ice_only - row[3] += path in ice_only and path in hit_paths - logged = [0, 0] - for site, (krate, path, _, _, _, panics) in blocks.items(): - if krate in NOT_AT_RUN_TIME or path in unreach or path not in paths: - continue - if site in logging: - logged[0] += 1 - logged[1] += site in hit - continue - row = rows[krate] - row[0] += 1 - row[1] += site in hit - panics = panics or path in ice_only - row[2] += panics - row[3] += panics and site in hit - n, r, pn, pr = (sum(row[i] for row in rows.values()) for i in range(4)) - print(f"blocks: of the {n} in reachable functions, {r} ran ({100 * r / max(n, 1):.1f}%); " - f"{pn} only panic ({pr} ran): without them, {r - pr} of {n - pn} " - f"({100 * (r - pr) / max(n - pn, 1):.1f}%); not counted: {logged[0]} blocks of logging " - f"macros, which run only with RUSTC_LOG ({logged[1]} ran)") - print(f"{'crate':40} {'ran':>7} {'blocks':>9} {'%':>6} (panic-only blocks aside)") - for krate, (n, r, pn, pr) in sorted(rows.items(), key=lambda kv: (kv[1][1] - kv[1][3]) / max(kv[1][0] - kv[1][2], 1)): - if n - pn: - print(f"{krate:40} {r - pr:7} {n - pn:9} {100 * (r - pr) / (n - pn):6.1f}") - print(f"{'crate':40} {'ran':>7} {'reachable':>9} {'%':>6}") - for krate, (n, r, _) in sorted(by_crate.items(), key=lambda kv: kv[1][1] / max(kv[1][0], 1)): - if n: - print(f"{krate:40} {r:7} {n:9} {100 * r / n:6.1f}") -for crate in args.unreachable: - print(f"\nunreachable in {crate}:") - for path in sorted(unreach): - if path.startswith(crate + "::"): - print(" ", path) -if args.json: - Path(args.json).write_text(json.dumps({"unreachable": sorted(unreach), - "ice_only": sorted(ice_only & paths), - "ran_unreachable": sorted(hit_paths & unreach), - "reachable_not_hit": sorted((paths - unreach) - hit_paths)}, indent=0)) - -if args.gaps: - files = defaultdict(list) - for site, (krate, path) in functions.items(): - if path not in unreach and path not in hit_paths: - span = span_of.get(path, "?") - files[(krate, span.rsplit(":", 2)[0])].append((span, path)) - with open(args.gaps, "w") as out: - out.write(f"# Reachable functions that never ran: {sum(map(len, files.values()))}\n\n") - by_crate = defaultdict(int) - for (krate, _), fs in files.items(): - by_crate[krate] += len(fs) - for krate in sorted(by_crate, key=lambda k: -by_crate[k]): - out.write(f"## {krate} ({by_crate[krate]})\n\n") - for (k, file), fs in sorted(files.items(), key=lambda kv: -len(kv[1])): - if k != krate: - continue - out.write(f"### {file} ({len(fs)})\n\n") - for span, path in sorted(fs): - tag = " (only panics)" if path in ice_only else "" - out.write(f"- `{path}` {span.rsplit(':', 2)[-2] if ':' in span else ''}{tag}\n") - out.write("\n") - print(f"gaps written to {args.gaps}") - -if args.block_gaps and blocks: - # The blocks that never ran, in functions that did (gaps.md has the functions that did not), - # by crate and file, the lines with the most first. - files = defaultdict(list) - for site, (krate, path, name, span, snippet, panics) in blocks.items(): - if path in hit_paths and site not in hit and krate not in NOT_AT_RUN_TIME and site not in logging: - files[(krate, span.rsplit(":", 2)[0])].append((span, path, name, snippet, panics or path in ice_only)) - with open(args.block_gaps, "w") as out: - out.write(f"# Blocks that never ran in functions that did: {sum(map(len, files.values()))}\n\n") - by_crate = defaultdict(int) - for (krate, _), bs in files.items(): - by_crate[krate] += len(bs) - for krate in sorted(by_crate, key=lambda k: -by_crate[k]): - out.write(f"## {krate} ({by_crate[krate]})\n\n") - for (k, file), bs in sorted(files.items(), key=lambda kv: -len(kv[1])): - if k != krate: - continue - out.write(f"### {file} ({len(bs)})\n\n") - def line_of(b): - parts = b[0].rsplit(":", 2) - return (int(parts[1]), int(parts[2])) if len(parts) == 3 and parts[1].isdigit() else (0, 0) - for b in sorted(bs, key=line_of): - span, path, name, snippet, panics = b - tag = " (only panics)" if panics else "" - out.write(f"- {line_of(b)[0]} `{path}` {name}{tag}: `{snippet[:100]}`\n") - out.write("\n") - print(f"block gaps written to {args.block_gaps}") diff --git a/rustc/coverage-report.sh b/rustc/coverage-report.sh index 879950a..0a8a644 100755 --- a/rustc/coverage-report.sh +++ b/rustc/coverage-report.sh @@ -3,7 +3,7 @@ # and option-configuration logs, against the call graph's denominator. Writes # $WORK/coverage-report.txt, $WORK/gaps.md (what can run and did not, by crate and file) and # $WORK/gaps.json; with a compiler built with `[coverage] blocks`, also $WORK/gaps-blocks.md (the -# blocks that never ran in functions that did). Extra arguments go to callgraph.py (`--why `, `--unreachable `). +# blocks that never ran in functions that did). Extra arguments go to `mirth-lab callgraph` (`--why `, `--unreachable `). # # WORK=~/mirth-work rustc/coverage-report.sh [--why ...] # @@ -34,7 +34,7 @@ dirs=${COV_LOGS-$work/cov-sink $work/cov-flags} for d in $dirs; do [ -d "$d" ] && logs+=(--logs "$d") done -python3 "$here/callgraph.py" --graph "$work/build-cg/mirth-sites" --sites "$build/mirth-sites" \ +"$here/../target/release/mirth-lab" callgraph --graph "$work/build-cg/mirth-sites" --sites "$build/mirth-sites" \ "${runs[@]}" "${logs[@]}" --gaps "$report/gaps.md" --block-gaps "$report/gaps-blocks.md" --json "$report/gaps.json" "$@" \ > "$report/coverage-report.txt" sed -n '1,5p;/^blocks:/p' "$report/coverage-report.txt" From 40a3634966b1d778c94ccd0d902df13f6d866907 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 07:51:55 +0000 Subject: [PATCH 13/23] instr-check: executable-no-mangle-strip is expected under IR PGO (clang -fprofile-generate fails the same link) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- Cargo.lock | 1 + crates/mirth-lab/src/tools/instr_check.rs | 5 ++++- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/Cargo.lock b/Cargo.lock index 3851a8e..d985a42 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -321,6 +321,7 @@ version = "0.0.0" dependencies = [ "anyhow", "clap", + "libc", "mirth-rewrite", "rayon", "regex", diff --git a/crates/mirth-lab/src/tools/instr_check.rs b/crates/mirth-lab/src/tools/instr_check.rs index 91158e9..7645d74 100644 --- a/crates/mirth-lab/src/tools/instr_check.rs +++ b/crates/mirth-lab/src/tools/instr_check.rs @@ -191,7 +191,10 @@ pub fn run(args: Args) -> anyhow::Result { let tc = Toolchain { rustc: root.join("bin/rustc"), tools: root.join("lib/rustlib/x86_64-unknown-linux-gnu/bin") }; // Threaded tests' output order depends on scheduling, which instrumentation changes. // Output that depends on a random hash seed. - const NOISE: &[&str] = &["collections/hashmap/hashmap-debug-format.rs"]; + // hashmap-debug-format: hash order. executable-no-mangle-strip: expects an unreferenced + // `#[no_mangle]` function (naming an undefined symbol) to be collected; IR PGO's profile data + // keeps every instrumented function, as clang -fprofile-generate does with --gc-sections. + const NOISE: &[&str] = &["collections/hashmap/hashmap-debug-format.rs", "linking/executable-no-mangle-strip.rs"]; let tests = uitest::tests(&args.sweep.tests, &[Some(Kind::RunPass)], |t| { uitest::flag_matches(t, &OWN) || THREADS.is_match(&t.text) || NOISE.contains(&t.rel.as_str()) }); From 5652f80bf8d9fc7299729a6293e15a5eb0a7bd89 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:01:01 +0000 Subject: [PATCH 14/23] mirth-lab ui-fuzz, and the mutations and Cargo-artifact comparison in the library; ui-solver-diff.py removed (solver-diff covers it) ui-fuzz: same options, outputs and finding classes as ui-fuzz.py on 48 tests (276 vs 283 edits; debuginfo and io-checks P5s; finding 18 reproduced on pinned-drop-sugar-no-core). Edits come from rand's StdRng seeded by the test path, so sequences differ from Python's random.Random. Unified diffs of the edit history from similar. mutations: the 16 edits of mutations.py; the lookbehinds done by hand. artifacts: collect and compare from artifacts.py (Collected), for the fuzz and replay ports. mutations.py and artifacts.py stay until fuzz.py, replay.py, flag-walk.py and coverage-flags.py are ported. ui-solver-diff.py: solver-diff on tests/ui/transmutability compiles the same 78 tests and finds the same 8 verdict differences; it notes the E0277/E0521 case. Not covered: a different first-error message under the same error code. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- Cargo.lock | 104 ++++++++- crates/mirth-lab/Cargo.toml | 2 + crates/mirth-lab/src/artifacts.rs | 94 ++++++++ crates/mirth-lab/src/lib.rs | 1 + crates/mirth-lab/src/main.rs | 4 + crates/mirth-lab/src/mutations.rs | 306 ++++++++++++++++++++++++++ crates/mirth-lab/src/tools/ui_fuzz.rs | 280 +++++++++++++++++++++++ docs/coverage-handoff.md | 2 +- docs/coverage.md | 6 +- docs/hunt.md | 2 +- docs/hunt/fatal-error-order.md | 2 +- docs/hunt/unleash-warning-lost.md | 2 +- rustc/coverage-run.sh | 2 +- rustc/ui-fuzz.py | 215 ------------------ rustc/ui-solver-diff.py | 102 --------- 15 files changed, 797 insertions(+), 327 deletions(-) create mode 100644 crates/mirth-lab/src/mutations.rs create mode 100644 crates/mirth-lab/src/tools/ui_fuzz.rs delete mode 100644 rustc/ui-fuzz.py delete mode 100644 rustc/ui-solver-diff.py diff --git a/Cargo.lock b/Cargo.lock index 3851a8e..c1c7bc2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -234,6 +234,18 @@ dependencies = [ "version_check", ] +[[package]] +name = "getrandom" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" +dependencies = [ + "cfg-if", + "libc", + "r-efi 5.3.0", + "wasip2", +] + [[package]] name = "getrandom" version = "0.4.3" @@ -242,7 +254,7 @@ checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" dependencies = [ "cfg-if", "libc", - "r-efi", + "r-efi 6.0.0", ] [[package]] @@ -321,12 +333,15 @@ version = "0.0.0" dependencies = [ "anyhow", "clap", + "libc", "mirth-rewrite", + "rand", "rayon", "regex", "serde", "serde_json", "sha2", + "similar", "tempfile", "wait-timeout", "walkdir", @@ -367,6 +382,15 @@ version = "1.70.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" +[[package]] +name = "ppv-lite86" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +dependencies = [ + "zerocopy", +] + [[package]] name = "proc-macro2" version = "1.0.107" @@ -385,12 +409,47 @@ dependencies = [ "proc-macro2", ] +[[package]] +name = "r-efi" +version = "5.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" + [[package]] name = "r-efi" version = "6.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" +[[package]] +name = "rand" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" +dependencies = [ + "rand_chacha", + "rand_core", +] + +[[package]] +name = "rand_chacha" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" +dependencies = [ + "ppv-lite86", + "rand_core", +] + +[[package]] +name = "rand_core" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" +dependencies = [ + "getrandom 0.3.4", +] + [[package]] name = "rayon" version = "1.12.0" @@ -525,6 +584,12 @@ dependencies = [ "digest", ] +[[package]] +name = "similar" +version = "2.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbbb5d9659141646ae647b42fe094daf6c6192d1620870b449d9557f748b2daa" + [[package]] name = "strsim" version = "0.11.1" @@ -560,7 +625,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" dependencies = [ "fastrand", - "getrandom", + "getrandom 0.4.3", "once_cell", "rustix", "windows-sys", @@ -654,6 +719,15 @@ dependencies = [ "winapi-util", ] +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen", +] + [[package]] name = "winapi-util" version = "0.1.11" @@ -690,6 +764,32 @@ version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" + +[[package]] +name = "zerocopy" +version = "0.8.61" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "879fb705ce98c32e41ebdb970fbe1204f8492423b314c6ab0354c3e7b5542866" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.61" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "708882a28301d604fa039cc7727607a98d04c4b86dc76ec9cc683f805d709759" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "zmij" version = "1.0.23" diff --git a/crates/mirth-lab/Cargo.toml b/crates/mirth-lab/Cargo.toml index 83fe69a..c5b3c90 100644 --- a/crates/mirth-lab/Cargo.toml +++ b/crates/mirth-lab/Cargo.toml @@ -14,9 +14,11 @@ path = "src/main.rs" anyhow = "1" clap = { version = "4.6", features = ["derive"] } rayon = "1.12" +rand = "0.9" regex = "1" serde = { version = "1", features = ["derive"] } serde_json = "1" +similar = "2" sha2 = "0.10" tempfile = "3" walkdir = "2" diff --git a/crates/mirth-lab/src/artifacts.rs b/crates/mirth-lab/src/artifacts.rs index e64e9e0..431b0a0 100644 --- a/crates/mirth-lab/src/artifacts.rs +++ b/crates/mirth-lab/src/artifacts.rs @@ -89,6 +89,91 @@ pub fn digest_dir(dir: &Path) -> BTreeMap { out } +/// What a Cargo build produced for packages built from a path, from the JSON messages of +/// `cargo build --message-format=json-render-diagnostics`: every .rmeta and executable by +/// digest, every .rlib member by member, every rendered diagnostic counted per crate. Paths are +/// relative to the target directory. +#[derive(Default, Debug, PartialEq, serde::Serialize)] +pub struct Collected { + pub rmeta: BTreeMap, + pub rlib: BTreeMap>, + pub exe: BTreeMap, + pub diag: BTreeMap<(String, String), usize>, +} + +pub fn collect(stdout: &str, target: &Path) -> Collected { + let mut found = Collected::default(); + let rel = |f: &str| Path::new(f).strip_prefix(target).map_or(f.to_owned(), |r| r.display().to_string()); + for line in stdout.lines() { + let Ok(msg) = serde_json::from_str::(line) else { continue }; + if !msg["package_id"].as_str().unwrap_or("").contains("path+file") { + continue; + } + match msg["reason"].as_str() { + Some("compiler-message") => { + let m = &msg["message"]; + let text = m["rendered"].as_str().filter(|t| !t.is_empty()).or(m["message"].as_str()).unwrap_or(""); + let krate = msg["target"]["name"].as_str().unwrap_or("").to_owned(); + *found.diag.entry((krate, text.to_owned())).or_default() += 1; + } + Some("compiler-artifact") => { + for f in msg["filenames"].as_array().into_iter().flatten().filter_map(|f| f.as_str()) { + if f.ends_with(".rmeta") { + found.rmeta.insert(rel(f), sha256(&std::fs::read(f).unwrap_or_default())); + } else if f.ends_with(".rlib") { + found.rlib.insert(rel(f), normalized_rlib(Path::new(f))); + } + } + if let Some(f) = msg["executable"].as_str() { + let data = std::fs::read(f).unwrap_or_default(); + found.exe.insert(rel(f), sha256(&SESSION.replace_all(&data, &b"$1"[..]))); + } + } + _ => {} + } + } + found +} + +/// {kind: [what differs]} for the kinds that differ between two collections. +pub fn compare(a: &Collected, b: &Collected) -> BTreeMap<&'static str, Vec> { + fn keys<'a, V: PartialEq>(a: &'a BTreeMap, b: &'a BTreeMap) -> Vec { + let all: std::collections::BTreeSet<&String> = a.keys().chain(b.keys()).collect(); + all.into_iter().filter(|k| a.get(*k) != b.get(*k)).cloned().collect() + } + let mut out = BTreeMap::new(); + for (kind, x, y) in [("rmeta", &a.rmeta, &b.rmeta), ("exe", &a.exe, &b.exe)] { + let diff = keys(x, y); + if !diff.is_empty() { + out.insert(kind, diff); + } + } + let empty = BTreeMap::new(); + let rlibs: Vec = keys(&a.rlib, &b.rlib) + .into_iter() + .map(|rel| { + let members = keys(a.rlib.get(&rel).unwrap_or(&empty), b.rlib.get(&rel).unwrap_or(&empty)); + let more = if members.len() > 5 { " …" } else { "" }; + format!("{rel}: {}{more}", members[..members.len().min(5)].join(", ")) + }) + .collect(); + if !rlibs.is_empty() { + out.insert("rlib", rlibs); + } + if a.diag != b.diag { + let only = |x: &BTreeMap<(String, String), usize>, y: &BTreeMap<(String, String), usize>, which: &str| -> Vec { + x.iter() + .filter(|(k, n)| **n > y.get(*k).copied().unwrap_or(0)) + .map(|((c, t), _)| format!("{c} only in the {which}: {:?}", t.chars().take(200).collect::())) + .collect() + }; + let mut d = only(&a.diag, &b.diag, "first"); + d.extend(only(&b.diag, &a.diag, "second")); + out.insert("diag", d); + } + out +} + #[cfg(test)] mod tests { use super::*; @@ -102,4 +187,13 @@ mod tests { let m = ar_members(&ar); assert!(m.contains_key("t.rcgu.o"), "{m:?}"); } + + #[test] + fn diagnostics_compared_by_count() { + let line = |t: &str| format!(r#"{{"reason":"compiler-message","package_id":"path+file:///x#a@0.1.0","target":{{"name":"a"}},"message":{{"rendered":"{t}"}}}}"#); + let a = collect(&format!("{}\n{}", line("w"), line("w")), Path::new("/t")); + let b = collect(&line("w"), Path::new("/t")); + assert_eq!(compare(&a, &b)["diag"], vec![r#"a only in the first: "w""#]); + assert!(compare(&a, &a).is_empty()); + } } diff --git a/crates/mirth-lab/src/lib.rs b/crates/mirth-lab/src/lib.rs index 0d8d916..bb2a8ba 100644 --- a/crates/mirth-lab/src/lib.rs +++ b/crates/mirth-lab/src/lib.rs @@ -4,6 +4,7 @@ pub mod artifacts; pub mod driver; pub mod miri; +pub mod mutations; pub mod normalize; pub mod rustc; pub mod uitest; diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs index c4dbcea..e5c901b 100644 --- a/crates/mirth-lab/src/main.rs +++ b/crates/mirth-lab/src/main.rs @@ -20,6 +20,7 @@ mod tools { pub mod scale_check; pub mod solver_diff; pub mod suggest_diff; + pub mod ui_fuzz; pub mod xlink; } @@ -60,6 +61,8 @@ enum Check { ScaleCheck(tools::scale_check::Args), /// rustc's extern "C" lowering matches clang's for random C signatures, per target. AbiDiff(tools::abi_diff::Args), + /// Incremental rebuilds of UI tests after random edits match clean builds. + UiFuzz(tools::ui_fuzz::Args), /// Reachability over the compiler's call graph; coverage of what can run, and gap lists. Callgraph(tools::callgraph::Args), } @@ -81,6 +84,7 @@ fn main() -> ExitCode { Check::Xlink(a) => tools::xlink::run(a), Check::ScaleCheck(a) => tools::scale_check::run(a), Check::AbiDiff(a) => tools::abi_diff::run(a), + Check::UiFuzz(a) => tools::ui_fuzz::run(a), Check::Callgraph(a) => tools::callgraph::run(a), }; match result { diff --git a/crates/mirth-lab/src/mutations.rs b/crates/mirth-lab/src/mutations.rs new file mode 100644 index 0000000..46749e5 --- /dev/null +++ b/crates/mirth-lab/src/mutations.rs @@ -0,0 +1,306 @@ +//! Random mechanical edits to Rust source, shared by the fuzzers. +//! +//! Each edit takes the file's text, a random generator and a counter, and returns the new text, +//! or None when it does not apply. EDITS lists them with weights. + +use std::sync::LazyLock; + +use rand::distr::Distribution; +use rand::distr::weighted::WeightedIndex; +use rand::rngs::StdRng; +use rand::seq::{IndexedRandom, SliceRandom}; +use rand::{Rng, SeedableRng}; +use regex::Regex; + +pub type Edit = fn(&str, &mut StdRng, usize) -> Option; + +/// A generator seeded by a name (a test path), so a run can be repeated. +pub fn seeded(name: &str) -> StdRng { + let digest = crate::artifacts::sha256(name.as_bytes()); + StdRng::seed_from_u64(u64::from_str_radix(&digest[..16], 16).unwrap()) +} + +/// A random edit, by weight. +pub fn pick(rng: &mut StdRng) -> (&'static str, Edit) { + static WEIGHTS: LazyLock> = LazyLock::new(|| WeightedIndex::new(EDITS.iter().map(|e| e.2)).unwrap()); + let (name, f, _) = EDITS[WEIGHTS.sample(rng)]; + (name, f) +} + +static INDENT: LazyLock = LazyLock::new(|| Regex::new(r"^\s*").unwrap()); + +fn indent(line: &str) -> &str { + INDENT.find(line).map_or("", |m| m.as_str()) +} + +/// Top-level items: blank-line separated, continuation blocks merged. +pub fn blocks(text: &str) -> Vec { + let mut out: Vec = Vec::new(); + for b in text.split("\n\n") { + let continues = b.chars().next().is_some_and(char::is_whitespace) || b.starts_with('}') || b.starts_with("where"); + match out.last_mut() { + Some(last) if continues => { + last.push_str("\n\n"); + last.push_str(b); + } + _ => out.push(b.to_owned()), + } + } + out +} + +fn lines(text: &str) -> Vec { + text.split('\n').map(str::to_owned).collect() +} + +pub fn comment_line(text: &str, rng: &mut StdRng, n: usize) -> Option { + let mut l = lines(text); + let i = rng.random_range(0..=l.len()); + let ind = indent(l.get(i).map_or("", String::as_str)).to_owned(); + l.insert(i, format!("{ind}// fuzz {n}")); + Some(l.join("\n")) +} + +pub fn blank_line(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut l = lines(text); + let i = rng.random_range(0..=l.len()); + l.insert(i, String::new()); + Some(l.join("\n")) +} + +pub fn remove_comment(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut l = lines(text); + let idx: Vec = (0..l.len()) + .filter(|&i| { + let t = l[i].trim(); + t.starts_with("//") && !t.starts_with("//!") + }) + .collect(); + let i = *idx.choose(rng)?; + l.remove(i); + Some(l.join("\n")) +} + +pub fn indent_line(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut l = lines(text); + let idx: Vec = (0..l.len()).filter(|&i| !l[i].trim().is_empty()).collect(); + let i = *idx.choose(rng)?; + l[i] = format!(" {}", l[i]); + Some(l.join("\n")) +} + +pub fn swap_items(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + let i = rng.random_range(1..b.len() - 1); + b.swap(i, i + 1); + Some(b.join("\n\n")) +} + +pub fn move_item_to_end(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + let item = b.remove(rng.random_range(1..b.len())); + b.push(item.trim_end_matches('\n').to_owned()); + Some(b.join("\n\n") + "\n") +} + +pub fn delete_item(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + b.remove(rng.random_range(1..b.len())); + Some(b.join("\n\n")) +} + +static FN_ITEM: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^(pub(\([^)]*\))? )?(const )?(async )?fn \w+").unwrap()); +static FN_NAME: LazyLock = LazyLock::new(|| Regex::new(r"\bfn (\w+)").unwrap()); + +pub fn duplicate_fn(text: &str, rng: &mut StdRng, n: usize) -> Option { + let mut b = blocks(text); + let fns: Vec = (0..b.len()).filter(|&i| FN_ITEM.is_match(&b[i])).collect(); + let i = *fns.choose(rng)?; + let copy = FN_NAME.replacen(&b[i], 1, format!("fn ${{1}}_fuzz{n}")).into_owned(); + b.insert(i + 1, copy); + Some(b.join("\n\n")) +} + +const ADDITIONS: &[&str] = &[ + "fn fuzz_private_{n}() -> u32 { {n} }", + "pub fn fuzz_public_{n}(x: u32) -> u32 { x.wrapping_mul({n}) }", + "#[inline]\npub fn fuzz_inline_{n}(x: &T) -> (T, u32) { (x.clone(), {n}) }", + "pub const FUZZ_{n}: &str = \"fuzz {n}\";", + "pub static FUZZ_STATIC_{n}: [u8; 3] = [{n} as u8, 1, 2];", + "#[derive(Debug, Clone, PartialEq)]\npub struct Fuzz{n} { pub items: [T; N], pub tag: &'static str }", + "pub enum FuzzEnum{n} { A(u32), B { x: i64 }, C }", + "pub trait FuzzTrait{n} { fn go(&self) -> impl Sized; const K: u32 = {n}; }", + "pub async fn fuzz_async_{n}() -> u32 { {n} }", + "pub type FuzzAlias{n} = Vec<(T, u32)>;", + "macro_rules! fuzz_macro_{n} { ($e:expr) => { $e + {n} }; }", + "pub mod fuzz_mod_{n} { pub fn inner() -> &'static str { \"{n}\" } }", +]; + +pub fn add_item(text: &str, rng: &mut StdRng, n: usize) -> Option { + let mut b = blocks(text); + let i = rng.random_range(1..=b.len()); + b.insert(i, ADDITIONS.choose(rng).unwrap().replace("{n}", &n.to_string())); + Some(b.join("\n\n")) +} + +fn word_or(c: Option, extra: char) -> bool { + c.is_some_and(|c| c.is_alphanumeric() || c == '_' || c == extra) +} + +static DIGITS: LazyLock = LazyLock::new(|| Regex::new(r"\d+").unwrap()); + +pub fn int_literal(text: &str, rng: &mut StdRng, _: usize) -> Option { + // A whole run of digits, not touching a word character or a dot on either side. + let ms: Vec = DIGITS + .find_iter(text) + .filter(|m| !word_or(text[..m.start()].chars().next_back(), '.') && !word_or(text[m.end()..].chars().next(), '.')) + .collect(); + let m = ms.choose(rng)?; + let bumped = m.as_str().parse::().map_or_else(|_| format!("{}1", m.as_str()), |v| (v + 1).to_string()); + Some(format!("{}{bumped}{}", &text[..m.start()], &text[m.end()..])) +} + +static STRING: LazyLock = LazyLock::new(|| Regex::new(r#""([^"\\\n]*)""#).unwrap()); + +pub fn str_literal(text: &str, rng: &mut StdRng, _: usize) -> Option { + // A string literal not preceded by a word character or a backslash; after a rejected + // opening quote the search resumes one character later, as a lookbehind would. + let mut ms = Vec::new(); + let mut pos = 0; + while let Some(c) = STRING.captures_at(text, pos) { + let m = c.get(0).unwrap(); + if word_or(text[..m.start()].chars().next_back(), '\\') { + pos = m.start() + 1; + } else { + ms.push((m.start(), m.end(), c[1].to_owned())); + pos = m.end(); + } + } + let (s, e, body) = ms.choose(rng)?; + Some(format!("{}\"{body}~\"{}", &text[..*s], &text[*e..])) +} + +static FN_LINE: LazyLock = LazyLock::new(|| Regex::new(r"^\s*(pub(\([^)]*\))? )?(const )?fn ").unwrap()); + +pub fn toggle_inline(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut l = lines(text); + let inl: Vec = (0..l.len()).filter(|&i| matches!(l[i].trim(), "#[inline]" | "#[inline(never)]" | "#[inline(always)]")).collect(); + let fns: Vec = (0..l.len()).filter(|&i| FN_LINE.is_match(&l[i])).collect(); + if !inl.is_empty() && rng.random::() < 0.5 { + let i = *inl.choose(rng).unwrap(); + l.remove(i); + } else if !fns.is_empty() { + let i = *fns.choose(rng).unwrap(); + let attr = ["#[inline]", "#[inline(never)]", "#[cold]", "#[must_use]"].choose(rng).unwrap(); + let ind = indent(&l[i]).to_owned(); + l.insert(i, format!("{ind}{attr}")); + } else { + return None; + } + Some(l.join("\n")) +} + +static ITEM_LINE: LazyLock = + LazyLock::new(|| Regex::new(r"^\s*(pub(\([^)]*\))? )?(fn|struct|enum|trait|const|static|type|mod) ").unwrap()); + +pub fn doc_comment(text: &str, rng: &mut StdRng, n: usize) -> Option { + let mut l = lines(text); + let idx: Vec = (0..l.len()).filter(|&i| ITEM_LINE.is_match(&l[i])).collect(); + let i = *idx.choose(rng)?; + let ind = indent(&l[i]).to_owned(); + l.insert(i, format!("{ind}/// Fuzz doc {n}, see [`Vec`].")); + Some(l.join("\n")) +} + +static DERIVE: LazyLock = LazyLock::new(|| Regex::new(r"#\[derive\(([^)]*)\)\]").unwrap()); + +pub fn reorder_derive(text: &str, rng: &mut StdRng, _: usize) -> Option { + let ms: Vec = DERIVE.captures_iter(text).collect(); + let c = ms.choose(rng)?; + let mut names: Vec<&str> = c[1].split(',').map(str::trim).filter(|x| !x.is_empty()).collect(); + if names.len() < 2 { + return None; + } + names.shuffle(rng); + let m = c.get(0).unwrap(); + Some(format!("{}#[derive({})]{}", &text[..m.start()], names.join(", "), &text[m.end()..])) +} + +static PUB_ITEM: LazyLock = LazyLock::new(|| Regex::new(r"\bpub (fn|struct|enum|const|static|trait|mod|type) ").unwrap()); + +pub fn narrow_visibility(text: &str, rng: &mut StdRng, _: usize) -> Option { + let ms: Vec = PUB_ITEM.captures_iter(text).collect(); + let c = ms.choose(rng)?; + let m = c.get(0).unwrap(); + Some(format!("{}pub(crate) {} {}", &text[..m.start()], &c[1], &text[m.end()..])) +} + +static LET: LazyLock = LazyLock::new(|| Regex::new(r"\blet (mut )?([a-z_][a-z0-9_]*)\b").unwrap()); + +pub fn rename_local(text: &str, rng: &mut StdRng, n: usize) -> Option { + let ms: Vec = LET.captures_iter(text).collect(); + let c = ms.choose(rng)?; + let name = &c[2]; + if name == "_" { + return None; + } + let m = c.get(0).unwrap(); + // Rename from the binding to the end of the enclosing top-level block. + let end = text[m.end()..].find("\n}\n").map_or(text.len(), |i| m.end() + i); + let re = Regex::new(&format!(r"\b{}\b", regex::escape(name))).unwrap(); + let body = re.replace_all(&text[m.start()..end], format!("{name}_f{n}").as_str()).into_owned(); + Some(format!("{}{body}{}", &text[..m.start()], &text[end..])) +} + +pub const EDITS: &[(&str, Edit, u32)] = &[ + ("comment_line", comment_line, 10), + ("blank_line", blank_line, 6), + ("remove_comment", remove_comment, 3), + ("indent_line", indent_line, 4), + ("swap_items", swap_items, 6), + ("move_item_to_end", move_item_to_end, 3), + ("delete_item", delete_item, 2), + ("duplicate_fn", duplicate_fn, 4), + ("add_item", add_item, 8), + ("int_literal", int_literal, 6), + ("str_literal", str_literal, 4), + ("toggle_inline", toggle_inline, 4), + ("doc_comment", doc_comment, 4), + ("reorder_derive", reorder_derive, 2), + ("narrow_visibility", narrow_visibility, 2), + ("rename_local", rename_local, 3), +]; + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn literals() { + let mut rng = seeded("t"); + assert_eq!(int_literal("let x = 12; y.0; a1", &mut rng, 0).unwrap(), "let x = 13; y.0; a1"); + assert_eq!(str_literal(r#"r"a" + "b""#, &mut rng, 0).unwrap(), r#"r"a" + "b~""#); + assert!(int_literal("x1 y.2", &mut rng, 0).is_none()); + } + + #[test] + fn blocks_merge_continuations() { + assert_eq!(blocks("fn a() {\n\n x\n}\n\nfn b() {}"), vec!["fn a() {\n\n x\n}", "fn b() {}"]); + } + + #[test] + fn rename() { + let mut rng = seeded("t"); + let t = "fn f() {\n let x = 1;\n x + 1\n}\nfn g() { x }\n"; + assert_eq!(rename_local(t, &mut rng, 3).unwrap(), "fn f() {\n let x_f3 = 1;\n x_f3 + 1\n}\nfn g() { x }\n"); + } +} diff --git a/crates/mirth-lab/src/tools/ui_fuzz.rs b/crates/mirth-lab/src/tools/ui_fuzz.rs new file mode 100644 index 0000000..8afa0ab --- /dev/null +++ b/crates/mirth-lab/src/tools/ui_fuzz.rs @@ -0,0 +1,280 @@ +//! Incremental rebuilds of rustc's UI tests, most of which fail to compile on purpose: the +//! fuzzer's edits applied to each test file, each rebuild compared with a clean build of the same +//! source, so that error reporting and recovery are exercised under incremental compilation. +//! +//! For each test in the list (ui-coverage's pick), compiled the way its `//@` headers say: build +//! it incrementally, then repeatedly apply a random edit (`mutations`) and rebuild it +//! incrementally, build the edited file again with a fresh incremental directory (same file, same +//! working directory), and compare: +//! +//! status both succeed, both fail, or both crash +//! diag the diagnostics, with paths and the incremental directory taken out +//! output the .rmeta and .rlib (normalized as `artifacts` does) when both succeed +//! ice both crash or neither does +//! +//! A clean build is made again before a difference counts (nondeterminism: such findings are +//! marked P5). Findings go to /findings/-/ with the source, both outputs and the +//! edit history. + +use std::collections::{BTreeMap, BTreeSet, HashSet}; +use std::io::Write as _; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::{LazyLock, Mutex}; +use std::time::Duration; + +use mirth_lab::artifacts::{normalized_rlib, sha256}; +use mirth_lab::rustc::{Exit, is_ice, run_command}; +use mirth_lab::uitest::{self, Kind}; +use mirth_lab::mutations; +use rayon::prelude::*; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[arg(long)] + tests: PathBuf, + /// The tests to fuzz: a JSON list of paths, or of objects with a "test" path. + #[arg(long)] + list: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 20)] + edits: usize, + #[arg(long, default_value_t = 8)] + jobs: usize, + /// Extra rustc options for every build. + #[arg(long, default_value = "", allow_hyphen_values = true)] + flags: String, + #[arg(long)] + pause_on_finding: bool, +} + +/// Tests that hit bugs already in docs/hunt.md every time: finding 17. +const KNOWN: &[&str] = &["unleash-the-miri-inside-of-you"]; + +#[derive(PartialEq)] +enum Digest { + File(String), + Rlib(BTreeMap), +} + +struct Build { + code: i32, + ice: bool, + diag: Vec, + files: BTreeMap, + stderr: String, +} + +static COUNT: LazyLock = LazyLock::new(|| Regex::new(r"\(\d+\)").unwrap()); +static NON_WORD: LazyLock = LazyLock::new(|| Regex::new(r"\W").unwrap()); + +struct Ctx<'a> { + args: &'a Args, + work: PathBuf, +} + +impl Ctx<'_> { + /// Compile `source` (a file in `dir`) with outputs in `dir/out`. + fn build(&self, dir: &Path, source: &Path, flags: &[String], edition: &str, kind: Option, incremental: Option<&Path>) -> Build { + let out = dir.join("out"); + let _ = std::fs::remove_dir_all(&out); + let _ = std::fs::create_dir_all(&out); + let emit = if Kind::is_check(kind) { "--emit=metadata" } else { "--emit=link,metadata" }; + let mut cmd = Command::new(&self.args.rustc); + cmd.arg(source.file_name().unwrap()) + .args(["--edition", edition, emit, "--out-dir", "out", "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"]) + .arg("--error-format=short"); + // Before the test's own flags, which may end with an option expecting a value. + if let Some(i) = incremental { + cmd.arg(format!("-Cincremental={}", i.display())); + } + cmd.args(flags).args(self.args.flags.split_whitespace()); + cmd.current_dir(dir).env("RUSTC_BOOTSTRAP", "1").env("RUST_BACKTRACE", "0"); + let (code, mut err) = match run_command(cmd, Duration::from_secs(300)) { + Ok(f) => match f.exit { + Exit::Code(c) => (c, f.stderr_text()), + Exit::Signal(s) => (-s, f.stderr_text()), + Exit::Timeout => (-1, "timeout".to_owned()), + }, + Err(e) => (-1, e.to_string()), + }; + let ice = is_ice(&err); + if let Some(i) = incremental { + err = err.replace(&i.display().to_string(), ""); + } + let diag: BTreeSet = err + .lines() + .filter(|l| !l.is_empty() && !["note: ", " ", "query stack", "#"].iter().any(|p| l.starts_with(p))) + .map(|l| COUNT.replace_all(l, "(…)").into_owned()) + .collect(); + let mut files = BTreeMap::new(); + for e in std::fs::read_dir(&out).into_iter().flatten().flatten() { + let p = e.path(); + let name = e.file_name().to_string_lossy().into_owned(); + match p.extension().and_then(|x| x.to_str()) { + Some("rmeta") => { + files.insert(name, Digest::File(sha256(&std::fs::read(&p).unwrap_or_default()))); + } + Some("rlib") => { + files.insert(name, Digest::Rlib(normalized_rlib(&p))); + } + _ => {} + } + } + Build { code, ice, diag: diag.into_iter().collect(), files, stderr: err } + } + + fn fuzz_one(&self, test: &str) -> anyhow::Result { + if self.work.join("PAUSED").exists() { + return Ok("not run".into()); + } + let path = self.args.tests.join(test); + let text = String::from_utf8_lossy(&std::fs::read(&path)?).into_owned(); + let (flags, edition, kind, _) = uitest::headers(&text); + let edition = edition.as_deref().unwrap_or("2015"); + if KNOWN.iter().any(|k| text.contains(k)) { + return Ok("skipped (known)".into()); + } + let name = format!("{}-{}", NON_WORD.replace_all(test, "_"), &sha256(test.as_bytes())[..8]); + let home = self.work.join("w").join(&name); + let _ = std::fs::remove_dir_all(&home); + // The clean build uses the same directory and file, with a fresh incremental directory, + // so that nothing but incremental state tells the two builds apart. + let dir = home.join("src"); + std::fs::create_dir_all(&dir)?; + let file_name = path.file_name().unwrap(); + let src = dir.join(file_name); + std::fs::write(&src, &text)?; + let (incr, incr_clean) = (home.join("incr"), home.join("incr-clean")); + let mut rng = mutations::seeded(test); + let mut history: Vec = Vec::new(); + self.build(&dir, &src, &flags, edition, kind, Some(&incr)); + let mut found_any = 0; + for n in 0..self.args.edits { + let old = std::fs::read_to_string(&src)?; + let (edit, f) = mutations::pick(&mut rng); + let Some(new) = f(&old, &mut rng, n).filter(|new| *new != old) else { continue }; + std::fs::write(&src, &new)?; + let diff = similar::TextDiff::from_lines(&old, &new).unified_diff().header("a", "b").to_string(); + history.push(serde_json::json!({"edit": edit, "diff": diff})); + let inc = self.build(&dir, &src, &flags, edition, kind, Some(&incr)); + let _ = std::fs::remove_dir_all(&incr_clean); + let clean = self.build(&dir, &src, &flags, edition, kind, Some(&incr_clean)); + let mut found = compare(&inc, &clean); + if !found.is_empty() && !found.iter().any(|f| f.starts_with("ICE")) { + let _ = std::fs::remove_dir_all(&incr_clean); + let again = self.build(&dir, &src, &flags, edition, kind, Some(&incr_clean)); + if !compare(&clean, &again).is_empty() { + found = found.into_iter().map(|f| format!("P5 {f}")).collect(); + } + } + if found.is_empty() { + continue; + } + found_any += 1; + let d = self.work.join("findings").join(format!("{name}-{n}")); + std::fs::create_dir_all(&d)?; + std::fs::write(d.join(file_name), &new)?; + let detail = serde_json::json!({"test": test, "found": found, "flags": flags, "edition": edition, + "kind": kind, "history": history}); + std::fs::write(d.join("finding.json"), to_json(&detail))?; + std::fs::write(d.join("inc.stderr"), &inc.stderr)?; + std::fs::write(d.join("clean.stderr"), &clean.stderr)?; + if self.args.pause_on_finding && !found.iter().all(|f| f.starts_with("P5") || f.starts_with("known")) { + std::fs::write(self.work.join("PAUSED"), to_json(&serde_json::json!({"test": test, "edit": n, "found": found})))?; + break; + } + } + let _ = std::fs::remove_dir_all(&home); + Ok(format!("{} edits, {found_any} findings", history.len())) + } +} + +/// JSON indented by one space, as the findings have always been written. +fn to_json(v: &impl Serialize) -> Vec { + let mut out = Vec::new(); + let mut ser = serde_json::Serializer::with_formatter(&mut out, serde_json::ser::PrettyFormatter::with_indent(b" ")); + v.serialize(&mut ser).expect("serializable"); + out +} + +fn compare(inc: &Build, clean: &Build) -> Vec { + let mut found = Vec::new(); + // Some tests crash the compiler on purpose; only a crash on one side counts. + if inc.ice != clean.ice { + found.push(format!("ICE {} only", if inc.ice { "incremental" } else { "clean" })); + } + if (inc.code == 0) != (clean.code == 0) { + found.push(format!("status: incremental {}, clean {}", inc.code, clean.code)); + } + let message = |l: &str| -> String { + let l = l.split_once(": error").map_or(l, |x| x.1); + l.split_once(": warning").map_or(l, |x| x.1).to_owned() + }; + let summary = |l: &str| l.starts_with("error: aborting") || (l.starts_with("warning:") && l.contains("emitted")); + if inc.diag != clean.diag { + let clean_set: HashSet<&String> = clean.diag.iter().collect(); + let inc_messages: HashSet = inc.diag.iter().map(|d| message(d)).collect(); + let fewer = inc.code != 0 + && clean.code != 0 + && inc.diag.iter().filter(|d| !summary(d)).all(|d| clean_set.contains(d)) + && clean.diag.iter().filter(|d| !inc.diag.contains(d)).all(|d| inc_messages.contains(&message(d)) || summary(d)); + if fewer { + found.push("known diag (finding 18): the rebuild stopped at a fatal error sooner".into()); + } else { + let only_inc: Vec<&String> = inc.diag.iter().filter(|d| !clean.diag.contains(d)).take(3).collect(); + let only_clean: Vec<&String> = clean.diag.iter().filter(|d| !inc.diag.contains(d)).take(3).collect(); + found.push(format!("diag: incremental only {only_inc:?}; clean only {only_clean:?}")); + } + } + if inc.code == 0 && clean.code == 0 && inc.files != clean.files { + let names: BTreeSet<&String> = inc.files.keys().chain(clean.files.keys()).collect(); + let differ: Vec<&str> = names.into_iter().filter(|k| inc.files.get(*k) != clean.files.get(*k)).map(String::as_str).collect(); + found.push(format!("output: {}", differ.join(", "))); + } + found +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let _ = std::fs::remove_file(work.join("PAUSED")); + let ctx = Ctx { args: &args, work: work.clone() }; + // A compiler without its standard library fails every test the same way: stop instead. + let probe_dir = work.join("probe"); + std::fs::create_dir_all(&probe_dir)?; + std::fs::write(probe_dir.join("probe.rs"), "fn main() {}\n")?; + let probe = ctx.build(&probe_dir, &probe_dir.join("probe.rs"), &[], "2021", Some(Kind::BuildPass), None); + if probe.code != 0 { + eprintln!("{} cannot build an empty program:\n{}", args.rustc.display(), probe.stderr); + return Ok(ExitCode::from(1)); + } + let picked: Vec = serde_json::from_slice(&std::fs::read(&args.list)?)?; + let tests: Vec = picked + .iter() + .filter_map(|t| t.get("test").unwrap_or(t).as_str().map(str::to_owned)) + .collect(); + let done_path = work.join("done.txt"); + let done: HashSet = std::fs::read_to_string(&done_path).unwrap_or_default().split_whitespace().map(str::to_owned).collect(); + let log = Mutex::new(std::fs::OpenOptions::new().create(true).append(true).open(&done_path)?); + let todo: Vec<&String> = tests.iter().filter(|t| !done.contains(*t)).collect(); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + pool.install(|| { + todo.par_iter().for_each(|test| { + // A broken test or harness case must not stop the run. + let result = ctx.fuzz_one(test).unwrap_or_else(|e| format!("harness error: {e:#}").chars().take(300).collect()); + println!("{test}: {result}"); + if !result.starts_with("not run") && !work.join("PAUSED").exists() { + let mut log = log.lock().unwrap(); + let _ = writeln!(log, "{test}"); + let _ = log.flush(); + } + }) + }); + Ok(ExitCode::SUCCESS) +} diff --git a/docs/coverage-handoff.md b/docs/coverage-handoff.md index f68b6fe..3135b31 100644 --- a/docs/coverage-handoff.md +++ b/docs/coverage-handoff.md @@ -65,7 +65,7 @@ $WORK` runs the ui-fulldeps tests compiletest skips at stage 1. **Any other command** (the generators, the fuzzers): ```sh -rustc/coverage-run.sh ui-fuzz python3 rustc/ui-fuzz.py --rustc $COV_RUSTC \ +rustc/coverage-run.sh ui-fuzz target/release/mirth-lab ui-fuzz --rustc $COV_RUSTC \ --tests $MIRTH_RUST/tests/ui --list $WORK/ui-cov/all-runnable.json --work $WORK/ui-fuzz-cov \ --edits 3 --jobs 6 # about 45 minutes rustc/coverage-run.sh generators python3 rustc/coverage-generators.py --rustc $COV_RUSTC \ diff --git a/docs/coverage.md b/docs/coverage.md index c6c9ee1..f3622da 100644 --- a/docs/coverage.md +++ b/docs/coverage.md @@ -77,7 +77,7 @@ reach 13,626 of them (55%)**. The first few: | `attributes/malformed-attrs.rs` | 444 | | `abi/stack-protector.rs` | 432 | -`rustc/ui-fuzz.py` runs the picked tests through incremental rebuilds after the fuzzer's edits, +`mirth-lab ui-fuzz` runs the picked tests through incremental rebuilds after the fuzzer's edits, each compared with a clean build (status, diagnostics, outputs): error reporting and recovery under incremental compilation, which sink cannot reach. @@ -86,7 +86,7 @@ solver, while nightly defaults to the new one: [`solver.md`](solver.md). ### Through incremental rebuilds -`rustc/ui-fuzz.py` over every UI test that compiles standalone (18,553 files), up to 8 random +`mirth-lab ui-fuzz` over every UI test that compiles standalone (18,553 files), up to 8 random edits each: **123,063 edit, incremental rebuild and clean rebuild cycles**, each compared on exit status, diagnostics and outputs. Checked first on finding 7's shape (a warning from inline assembly, lost when a codegen unit is reused), which it finds in 2 of 37 edits. @@ -201,7 +201,7 @@ logs the same way. | run | functions reached | |---|---:| -| ui tests through incremental rebuilds (`ui-fuzz.py`, 18,553 tests, 3 edits each) | 44,196 | +| ui tests through incremental rebuilds (`mirth-lab ui-fuzz`, 18,553 tests, 3 edits each) | 44,196 | | `coverage-generators.py`: every `--print` request on the host and on all 334 targets; minicore and an ABI file compiled for every target at `-Copt-level=0` and 3; the 300 picked UI tests under 48 debugging and printing options | 41,768 | | `coverage-generators.py --only links`: a binary, cdylib, staticlib and dylib on minicore for every target with `-Clinker=true`, under 11 sets of linker options | 18,228 | diff --git a/docs/hunt.md b/docs/hunt.md index 4a02cd5..5253ad6 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -41,7 +41,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 15 | under the incomplete `guard_patterns` feature, a guard pattern's guard is ignored: `Some(x if x > 3)` matches `Some(2)`, and the binding cannot be used in the arm | incomplete feature; found while covering nightly syntax in sink ([`grammar.md`](grammar.md)); [facts](hunt/guard-patterns-ignored.md), no fix | | 16 | with `-Zcache-proc-macros=yes -Zmetadata-crate-hash=no`, an incremental rebuild of a crate using derives gets a different crate hash (SVH) from a clean build after an edit upstream | unstable options (one "potentially unsound"); found by the fuzzer under walk configurations; not root-caused; [facts](hunt/cached-proc-macros-crate-hash.md); excluded from the models | | 17 | with `-Zunleash-the-miri-inside-of-you`, the "skipping const checks" warning is not shown again on an incremental rebuild | testing-only option; found by the UI-test fuzzer ([`coverage.md`](coverage.md)); since at least 1.60; [facts](hunt/unleash-warning-lost.md); tests using the option skipped | -| 18 | after a fatal error (a missing lang item), an incremental rebuild reports fewer errors than a clean build: the fatal error is reached in a different query order | diagnostics only; found by the UI-test fuzzer; stock nightly; [facts](hunt/fatal-error-order.md); labelled known in `ui-fuzz.py` | +| 18 | after a fatal error (a missing lang item), an incremental rebuild reports fewer errors than a clean build: the fatal error is reached in a different query order | diagnostics only; found by the UI-test fuzzer; stock nightly; [facts](hunt/fatal-error-order.md); labelled known in `mirth-lab ui-fuzz` | | 19 | on riscv64 and loongarch64, an `extern "C"` call passes an `i32` (or narrower integer) that lands on the stack without sign-extending it; a clang-compiled callee reads the slot as already extended | **looks new**; ABI, stable code; found by the ABI differential against clang ([`checks.md`](checks.md)); since at least 1.80; cause found (extension only `if *avail_gprs >= 1` in `callconv/riscv.rs`, same in `loongarch.rs`); [facts](hunt/riscv-stack-arg-extension.md) | | 20 | on RISC-V and LoongArch hard-float targets, a `repr(C)` struct of one float and one pointer is passed in a floating-point and an integer register; clang passes it by the integer convention, so C and Rust disagree on where it is | **looks new**; ABI, stable code; found by the ABI differential; since at least 1.80; cause found (`Primitive::Pointer` counted as an integer in `should_use_fp_conv_helper`, `callconv/riscv.rs` and `loongarch.rs`); [facts](hunt/riscv-float-pointer-struct.md) | | 21 | `-Zvalidate-mir` rejects MIR the compiler builds from accepted code: projections into `#[repr(simd)]` types (banned by MCP#838) in 9 SIMD tests, and an unsize coercion to `Pin>` in `async-await/issue-86507.rs` | found by the internal-checks sweep (`mirth-lab crash-diff`); stock nightly with `-Zvalidate-mir`; compiletest does not validate UI tests; [facts](hunt/internal-checks.md) | diff --git a/docs/hunt/fatal-error-order.md b/docs/hunt/fatal-error-order.md index 554ed20..5e8623b 100644 --- a/docs/hunt/fatal-error-order.md +++ b/docs/hunt/fatal-error-order.md @@ -21,6 +21,6 @@ rebuild, a query reached earlier (while checking what can be reused) hits the mi outside that loop, and the first fatal error ends the session. Any fatal error reached in a different order would do the same. -**How mirth found it.** `rustc/ui-fuzz.py` over rustc's UI tests, after an edit renamed the lang +**How mirth found it.** `mirth-lab ui-fuzz` over rustc's UI tests, after an edit renamed the lang item. Such differences (the rebuild's diagnostics a subset of the clean build's, each missing line repeating a message the rebuild has) are labelled known since. diff --git a/docs/hunt/unleash-warning-lost.md b/docs/hunt/unleash-warning-lost.md index 48b035a..f52c8ff 100644 --- a/docs/hunt/unleash-warning-lost.md +++ b/docs/hunt/unleash-warning-lost.md @@ -22,5 +22,5 @@ list at the end of the session. A rebuild that takes const checking's results fr incremental cache records nothing, so there is nothing to warn about: a side effect of a query that is not replayed, the same kind as finding 7. The option is for testing the compiler. -**How mirth found it.** `rustc/ui-fuzz.py` over rustc's UI tests ([`coverage.md`](../coverage.md)): +**How mirth found it.** `mirth-lab ui-fuzz` over rustc's UI tests ([`coverage.md`](../coverage.md)): the first edit to this test, its rebuild's diagnostics compared with a clean build's. diff --git a/rustc/coverage-run.sh b/rustc/coverage-run.sh index 77a09a5..24b403a 100755 --- a/rustc/coverage-run.sh +++ b/rustc/coverage-run.sh @@ -6,7 +6,7 @@ # WORK=~/mirth-work rustc/coverage-run.sh # # For example, the UI tests through incremental rebuilds: -# rustc/coverage-run.sh ui-fuzz python3 rustc/ui-fuzz.py --rustc $COV_RUSTC \ +# rustc/coverage-run.sh ui-fuzz target/release/mirth-lab ui-fuzz --rustc $COV_RUSTC \ # --tests $MIRTH_RUST/tests/ui --list ~/mirth-work/ui-cov/all-runnable.json \ # --work ~/mirth-work/ui-fuzz-cov --edits 3 --jobs 6 set -uo pipefail diff --git a/rustc/ui-fuzz.py b/rustc/ui-fuzz.py deleted file mode 100644 index 66ac3de..0000000 --- a/rustc/ui-fuzz.py +++ /dev/null @@ -1,215 +0,0 @@ -#!/usr/bin/env python3 -"""Incremental rebuilds of rustc's UI tests, most of which fail to compile on purpose: the -fuzzer's edits applied to each test file, each rebuild compared with a clean build of the same -source, so that error reporting and recovery are exercised under incremental compilation. - - rustc/ui-fuzz.py --rustc --tests /tests/ui --list picked.json --work

- [--edits 20] [--jobs 8] [--pause-on-finding] - -For each test in the list (rustc/ui-coverage.py pick), compiled the way its `//@` headers say: -build it incrementally, then repeatedly apply a random edit (mutations.py) and rebuild it -incrementally, build the edited file again with a fresh incremental directory (same file, same -working directory), and compare: - - status both succeed, both fail, or both crash - diag the diagnostics, with paths and the incremental directory taken out - output the .rmeta and .rlib (normalized as artifacts.py does) when both succeed - ice both crash or neither does - -A clean build is made again before a difference counts (nondeterminism). Findings go to -/findings/-/ with the source, both outputs and the edit history. -""" - -import argparse -import difflib -import hashlib -import json -import os -import random -import re -import shutil -import subprocess -import sys -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -here = Path(__file__).parent -sys.path.insert(0, str(here)) -import artifacts # noqa: E402 -import mutations # noqa: E402 - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--list", required=True) -p.add_argument("--work", required=True) -p.add_argument("--edits", type=int, default=20) -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--flags", default="", help="extra rustc options for every build") -p.add_argument("--pause-on-finding", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -NO_BUILD = ("check-pass", "check-fail") -# Tests that hit bugs already in docs/hunt.md every time: finding 17. -KNOWN = ("unleash-the-miri-inside-of-you",) - - -def headers(text): - flags, edition, revision, kind = [], None, None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - elif key in ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail"): - kind = key - if revision: - flags += ["--cfg", revision] - return flags, edition, kind - - -def build(directory, source, flags, edition, kind, incremental): - """Compile `source` (a file in `directory`) with outputs in `directory/out`.""" - out = directory / "out" - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - emit = "--emit=metadata" if kind in NO_BUILD or kind is None else "--emit=link,metadata" - argv = [args.rustc, source.name, "--edition", edition or "2015", emit, "--out-dir", "out", - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", "--error-format=short"] - # Before the test's own flags, which may end with an option expecting a value. - if incremental: - argv.append(f"-Cincremental={incremental}") - argv += [*flags, *args.flags.split()] - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=300, cwd=directory, - env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) - code, err = r.returncode, r.stderr - except subprocess.TimeoutExpired: - code, err = -1, "timeout" - ice = ("internal compiler error" in err or "the compiler unexpectedly panicked" in err - or "rustc interrupted by SIG" in err) - if incremental: - err = err.replace(incremental, "") - diag = sorted(set(re.sub(r"\(\d+\)", "(…)", l) for l in err.splitlines() - if l and not l.startswith(("note: ", " ", "query stack", "#")))) - files = {} - for f in sorted(out.iterdir()): - if f.suffix == ".rmeta": - files[f.name] = hashlib.sha256(f.read_bytes()).hexdigest() - elif f.suffix == ".rlib": - files[f.name] = artifacts.normalized_rlib(f) - return {"code": code, "ice": ice, "diag": diag, "files": files, "stderr": err} - - -def compare(inc, clean): - found = [] - # Some tests crash the compiler on purpose; only a crash on one side counts. - if inc["ice"] != clean["ice"]: - found.append("ICE " + ("incremental" if inc["ice"] else "clean") + " only") - if (inc["code"] == 0) != (clean["code"] == 0): - found.append(f"status: incremental {inc['code']}, clean {clean['code']}") - message = lambda line: line.split(": error", 1)[-1].split(": warning", 1)[-1] - summary = lambda line: line.startswith("error: aborting") or line.startswith("warning:") and "emitted" in line - fewer = (inc["code"] != 0 and clean["code"] != 0 - and {d for d in inc["diag"] if not summary(d)} <= set(clean["diag"]) - and all(message(d) in {message(e) for e in inc["diag"]} or summary(d) - for d in clean["diag"] if d not in inc["diag"])) - if inc["diag"] != clean["diag"] and fewer: - found.append("known diag (finding 18): the rebuild stopped at a fatal error sooner") - elif inc["diag"] != clean["diag"]: - only_inc = [d for d in inc["diag"] if d not in clean["diag"]][:3] - only_clean = [d for d in clean["diag"] if d not in inc["diag"]][:3] - found.append(f"diag: incremental only {only_inc}; clean only {only_clean}") - if inc["code"] == 0 and clean["code"] == 0 and inc["files"] != clean["files"]: - found.append("output: " + ", ".join(k for k in set(inc["files"]) | set(clean["files"]) - if inc["files"].get(k) != clean["files"].get(k))) - return found - - -def fuzz(test): - try: - return fuzz_one(test) - except Exception as error: # a broken test or harness case must not stop the run - return test, f"harness error: {error!r}"[:300] - - -def fuzz_one(test): - if (WORK / "PAUSED").exists(): - return test, "not run" - path = Path(args.tests) / test - text = path.read_text(errors="replace") - flags, edition, kind = headers(text) - if any(k in text for k in KNOWN): - return test, "skipped (known)" - name = re.sub(r"\W", "_", test) + "-" + hashlib.sha256(test.encode()).hexdigest()[:8] - home = WORK / "w" / name - shutil.rmtree(home, ignore_errors=True) - # The clean build uses the same directory and file, with a fresh incremental directory, so - # that nothing but incremental state tells the two builds apart. - inc_dir = clean_dir = home / "src" - inc_dir.mkdir(parents=True) - src_inc = src_clean = inc_dir / path.name - src_inc.write_text(text) - rng = random.Random(test) - history = [] - build(inc_dir, src_inc, flags, edition, kind, str(home / "incr")) - found_any = 0 - for n in range(args.edits): - old = src_inc.read_text() - fn = rng.choices([e for e, _ in mutations.EDITS], weights=[w for _, w in mutations.EDITS])[0] - new = fn(old, rng, n) - if new is None or new == old: - continue - src_inc.write_text(new) - history.append({"edit": fn.__name__, "diff": "".join(difflib.unified_diff( - old.splitlines(True), new.splitlines(True), "a", "b"))}) - inc = build(inc_dir, src_inc, flags, edition, kind, str(home / "incr")) - shutil.rmtree(home / "incr-clean", ignore_errors=True) - clean = build(clean_dir, src_clean, flags, edition, kind, str(home / "incr-clean")) - found = compare(inc, clean) - if found and not any(f.startswith("ICE") for f in found): - shutil.rmtree(home / "incr-clean", ignore_errors=True) - again = build(clean_dir, src_clean, flags, edition, kind, str(home / "incr-clean")) - if compare(clean, again): - found = ["P5 " + f for f in found] - if found: - found_any += 1 - d = WORK / "findings" / f"{name}-{n}" - d.mkdir(parents=True, exist_ok=True) - (d / path.name).write_text(new) - (d / "finding.json").write_text(json.dumps({"test": test, "found": found, "flags": flags, - "edition": edition, "kind": kind, - "history": history}, indent=1)) - (d / "inc.stderr").write_text(inc["stderr"]) - (d / "clean.stderr").write_text(clean["stderr"]) - if args.pause_on_finding and not all(f.startswith(("P5", "known")) for f in found): - (WORK / "PAUSED").write_text(json.dumps({"test": test, "edit": n, "found": found}, indent=1)) - break - shutil.rmtree(home, ignore_errors=True) - return test, f"{len(history)} edits, {found_any} findings" - - -WORK.mkdir(parents=True, exist_ok=True) -(WORK / "PAUSED").unlink(missing_ok=True) -# A compiler without its standard library fails every test the same way: stop instead. -(WORK / "probe").mkdir(exist_ok=True) -(WORK / "probe" / "probe.rs").write_text("fn main() {}\n") -probe = build(WORK / "probe", WORK / "probe" / "probe.rs", [], "2021", "build-pass", None) -if probe["code"] != 0: - sys.exit(f"{args.rustc} cannot build an empty program:\n{probe['stderr']}") -picked = json.loads(Path(args.list).read_text()) -tests = [t["test"] if isinstance(t, dict) else t for t in picked] -done_path = WORK / "done.txt" -done = set(done_path.read_text().split()) if done_path.exists() else set() -with ThreadPoolExecutor(args.jobs) as ex, done_path.open("a") as log: - for test, result in ex.map(fuzz, [t for t in tests if t not in done]): - print(f"{test}: {result}", flush=True) - if not result.startswith("not run") and not (WORK / "PAUSED").exists(): - log.write(test + "\n") - log.flush() diff --git a/rustc/ui-solver-diff.py b/rustc/ui-solver-diff.py deleted file mode 100644 index d4ee965..0000000 --- a/rustc/ui-solver-diff.py +++ /dev/null @@ -1,102 +0,0 @@ -#!/usr/bin/env python3 -"""rustc's UI tests under nightly's default trait solver and under the one the test suite pins. - -Nightly builds use the new trait solver everywhere by default (compiler-team MCP #1014, -rust-lang/rust#160895), but compiletest passes `-Znext-solver=coherence` to every UI test, so -the suite checks the old behaviour only. This compiles each test both ways, the way its `//@` -headers say (as rustc/ui-coverage.py does), and lists the tests whose outcome differs: an ICE, -success against failure, or a different first error. - - rustc/ui-solver-diff.py --rustc --tests /tests/ui --out [--jobs 8] - -Writes /results.jsonl and prints the differences, ICEs first. -""" - -import argparse -import json -import os -import re -import subprocess -import tempfile -from collections import Counter -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--out", required=True) -p.add_argument("--jobs", type=int, default=8) -args = p.parse_args() -OUT = Path(args.out) - -SKIP = re.compile(r"^//@\s*(aux-build|aux-crate|aux-bin|proc-macro|add-minicore|needs-llvm-components|" - r"only-(?!x86_64|linux|unix|64bit)|ignore-x86_64|ignore-linux|needs-sanitizer|needs-profiler|" - r"known-bug)|-Znext-solver", re.M) -NO_BUILD = ("check-pass", "check-fail") - - -def headers(text): - flags, edition, revision, kind = [], None, None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - elif key in ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail"): - kind = key - if revision: - flags += ["--cfg", revision] - return flags, edition, kind - - -def compile(path, flags, edition, kind, solver): - with tempfile.TemporaryDirectory(dir=OUT / "scratch") as d: - emit = "--emit=metadata" if kind in NO_BUILD or kind is None else "--emit=link" - argv = [args.rustc, str(path), "--edition", edition or "2015", emit, "--out-dir", d, - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", *solver, *flags] - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=120, cwd=d, - env=dict(os.environ, RUSTC_BOOTSTRAP="1")) - except subprocess.TimeoutExpired: - return {"status": "timeout", "error": ""} - ice = ("internal compiler error" in r.stderr or "the compiler unexpectedly panicked" in r.stderr - or "rustc interrupted by SIG" in r.stderr) - first = next((l for l in r.stderr.splitlines() if l.startswith("error")), "") - first = re.sub(r"/\S+|`[^`]*`", "…", first)[:160] - return {"status": "ok" if r.returncode == 0 else ("ice" if ice else "error"), "error": first} - - -def one(path): - text = path.read_text(errors="replace") - rel = str(path.relative_to(args.tests)) - if SKIP.search(text): - return {"test": rel, "skipped": True} - flags, edition, kind = headers(text) - default = compile(path, flags, edition, kind, []) - pinned = compile(path, flags, edition, kind, ["-Znext-solver=coherence"]) - return {"test": rel, "kind": kind, "default": default, "pinned": pinned} - - -OUT.mkdir(parents=True, exist_ok=True) -(OUT / "scratch").mkdir(exist_ok=True) -tests = sorted(p for p in Path(args.tests).rglob("*.rs") if "auxiliary" not in p.parts) -results = [] -with ThreadPoolExecutor(args.jobs) as ex, (OUT / "results.jsonl").open("w") as out: - for res in ex.map(one, tests): - out.write(json.dumps(res) + "\n") - results.append(res) - -ran = [r for r in results if not r.get("skipped")] -differ = [r for r in ran if r["default"]["status"] != r["pinned"]["status"] - or r["default"]["error"] != r["pinned"]["error"]] -print(f"{len(ran)} tests compiled both ways; {len(differ)} differ") -print(Counter((r["pinned"]["status"], r["default"]["status"]) for r in differ).most_common()) -for r in sorted(differ, key=lambda r: (r["default"]["status"] != "ice", r["test"])): - print(f"{r['pinned']['status']:>7} -> {r['default']['status']:<7} {r['test']} {r['default']['error'][:100]}") From fbfab7246b0bc46c8e69da65ffbf2f638d0e53b0 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:02:11 +0000 Subject: [PATCH 15/23] wip: flag tools --- Cargo.lock | 9 + crates/mirth-lab/Cargo.toml | 2 + crates/mirth-lab/src/artifacts.rs | 112 +++++ crates/mirth-lab/src/lib.rs | 1 + crates/mirth-lab/src/main.rs | 28 ++ crates/mirth-lab/src/mutations.rs | 328 +++++++++++++ crates/mirth-lab/src/tools/audit_options.rs | 160 ++++++ crates/mirth-lab/src/tools/flag_fuzz.rs | 92 ++++ crates/mirth-lab/src/tools/flag_min.rs | 163 +++++++ crates/mirth-lab/src/tools/flag_model.rs | 342 +++++++++++++ crates/mirth-lab/src/tools/flag_rows.rs | 71 +++ crates/mirth-lab/src/tools/flag_universe.rs | 368 ++++++++++++++ crates/mirth-lab/src/tools/flag_walk.rs | 515 ++++++++++++++++++++ 13 files changed, 2191 insertions(+) create mode 100644 crates/mirth-lab/src/mutations.rs create mode 100644 crates/mirth-lab/src/tools/audit_options.rs create mode 100644 crates/mirth-lab/src/tools/flag_fuzz.rs create mode 100644 crates/mirth-lab/src/tools/flag_min.rs create mode 100644 crates/mirth-lab/src/tools/flag_model.rs create mode 100644 crates/mirth-lab/src/tools/flag_rows.rs create mode 100644 crates/mirth-lab/src/tools/flag_universe.rs create mode 100644 crates/mirth-lab/src/tools/flag_walk.rs diff --git a/Cargo.lock b/Cargo.lock index 3851a8e..8ef484a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -321,12 +321,15 @@ version = "0.0.0" dependencies = [ "anyhow", "clap", + "fastrand", + "libc", "mirth-rewrite", "rayon", "regex", "serde", "serde_json", "sha2", + "similar", "tempfile", "wait-timeout", "walkdir", @@ -525,6 +528,12 @@ dependencies = [ "digest", ] +[[package]] +name = "similar" +version = "2.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbbb5d9659141646ae647b42fe094daf6c6192d1620870b449d9557f748b2daa" + [[package]] name = "strsim" version = "0.11.1" diff --git a/crates/mirth-lab/Cargo.toml b/crates/mirth-lab/Cargo.toml index 83fe69a..e2b88ae 100644 --- a/crates/mirth-lab/Cargo.toml +++ b/crates/mirth-lab/Cargo.toml @@ -12,6 +12,8 @@ path = "src/main.rs" [dependencies] anyhow = "1" +fastrand = "2" +similar = "2.7" clap = { version = "4.6", features = ["derive"] } rayon = "1.12" regex = "1" diff --git a/crates/mirth-lab/src/artifacts.rs b/crates/mirth-lab/src/artifacts.rs index e64e9e0..4470536 100644 --- a/crates/mirth-lab/src/artifacts.rs +++ b/crates/mirth-lab/src/artifacts.rs @@ -89,6 +89,118 @@ pub fn digest_dir(dir: &Path) -> BTreeMap { out } +/// What a Cargo build produced, for packages built from a path, in a form two builds can be +/// compared in (from `cargo build --message-format=json-render-diagnostics`). +#[derive(Default, Debug, Clone)] +pub struct Collected { + /// Every .rmeta, by path relative to the target directory. + pub rmeta: BTreeMap, + /// Every .rlib's members by name, session suffixes removed. + pub rlib: BTreeMap>, + /// Every executable. + pub exe: BTreeMap, + /// Every rendered diagnostic, counted per (target name, text). + pub diag: BTreeMap<(String, String), usize>, +} + +/// Artifacts from cargo's JSON messages, keyed by path relative to `target`. +pub fn collect(stdout: &str, target: &Path) -> Collected { + let mut found = Collected::default(); + let rel = |f: &str| -> String { + Path::new(f).strip_prefix(target).map_or_else(|_| f.to_owned(), |p| p.to_string_lossy().into_owned()) + }; + for line in stdout.lines() { + let Ok(msg) = serde_json::from_str::(line) else { continue }; + if !msg["package_id"].as_str().unwrap_or("").contains("path+file") { + continue; + } + let target_name = msg["target"]["name"].as_str().unwrap_or("").to_owned(); + match msg["reason"].as_str() { + Some("compiler-message") => { + let m = &msg["message"]; + let text = m["rendered"].as_str().filter(|s| !s.is_empty()).or(m["message"].as_str()).unwrap_or("").to_owned(); + *found.diag.entry((target_name, text)).or_default() += 1; + continue; + } + Some("compiler-artifact") => {} + _ => continue, + } + for f in msg["filenames"].as_array().into_iter().flatten().filter_map(|f| f.as_str()) { + if f.ends_with(".rmeta") { + found.rmeta.insert(rel(f), sha256(&std::fs::read(f).unwrap_or_default())); + } else if f.ends_with(".rlib") { + found.rlib.insert(rel(f), normalized_rlib(Path::new(f))); + } + } + if let Some(f) = msg["executable"].as_str() { + let data = std::fs::read(f).unwrap_or_default(); + found.exe.insert(rel(f), sha256(&SESSION.replace_all(&data, &b"$1"[..]))); + } + } + found +} + +/// Python's repr of a string, as the findings have always shown texts. +pub fn py_repr(s: &str) -> String { + let q = if s.contains('\'') && !s.contains('"') { '"' } else { '\'' }; + let mut out = String::from(q); + for c in s.chars() { + match c { + '\\' => out.push_str("\\\\"), + '\n' => out.push_str("\\n"), + '\r' => out.push_str("\\r"), + '\t' => out.push_str("\\t"), + c if c == q => { + out.push('\\'); + out.push(c); + } + c if (c as u32) < 0x20 || c as u32 == 0x7f => out.push_str(&format!("\\x{:02x}", c as u32)), + c => out.push(c), + } + } + out.push(q); + out +} + +/// {kind: [what differs]} for the kinds that differ between two collections. +pub fn compare(a: &Collected, b: &Collected) -> BTreeMap> { + let mut out = BTreeMap::new(); + for (kind, x, y) in [("rmeta", &a.rmeta, &b.rmeta), ("exe", &a.exe, &b.exe)] { + let keys: std::collections::BTreeSet<&String> = x.keys().chain(y.keys()).collect(); + let diff: Vec = keys.into_iter().filter(|k| x.get(*k) != y.get(*k)).cloned().collect(); + if !diff.is_empty() { + out.insert(kind.to_owned(), diff); + } + } + let empty = BTreeMap::new(); + let mut diff = Vec::new(); + let rels: std::collections::BTreeSet<&String> = a.rlib.keys().chain(b.rlib.keys()).collect(); + for rel in rels { + let (ma, mb) = (a.rlib.get(rel).unwrap_or(&empty), b.rlib.get(rel).unwrap_or(&empty)); + let members: std::collections::BTreeSet<&String> = ma.keys().chain(mb.keys()).filter(|m| ma.get(*m) != mb.get(*m)).collect(); + if !members.is_empty() { + let shown: Vec<&str> = members.iter().take(5).map(|m| m.as_str()).collect(); + diff.push(format!("{rel}: {}{}", shown.join(", "), if members.len() > 5 { " …" } else { "" })); + } + } + if !diff.is_empty() { + out.insert("rlib".to_owned(), diff); + } + if a.diag != b.diag { + let mut d = Vec::new(); + for (label, x, y) in [("first", &a.diag, &b.diag), ("second", &b.diag, &a.diag)] { + for ((c, t), n) in x { + if *n > y.get(&(c.clone(), t.clone())).copied().unwrap_or(0) { + let short: String = t.chars().take(200).collect(); + d.push(format!("{c} only in the {label}: {}", py_repr(&short))); + } + } + } + out.insert("diag".to_owned(), d); + } + out +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/mirth-lab/src/lib.rs b/crates/mirth-lab/src/lib.rs index 0d8d916..bb2a8ba 100644 --- a/crates/mirth-lab/src/lib.rs +++ b/crates/mirth-lab/src/lib.rs @@ -4,6 +4,7 @@ pub mod artifacts; pub mod driver; pub mod miri; +pub mod mutations; pub mod normalize; pub mod rustc; pub mod uitest; diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs index c4dbcea..18a8e0a 100644 --- a/crates/mirth-lab/src/main.rs +++ b/crates/mirth-lab/src/main.rs @@ -7,9 +7,16 @@ use clap::{Parser, Subcommand}; mod tools { pub mod abi_diff; + pub mod audit_options; pub mod callgraph; pub mod crash_diff; pub mod diag_check; + pub mod flag_fuzz; + pub mod flag_min; + pub mod flag_model; + pub mod flag_rows; + pub mod flag_universe; + pub mod flag_walk; pub mod gate_check; pub mod instr_check; pub mod miri_diff; @@ -62,6 +69,20 @@ enum Check { AbiDiff(tools::abi_diff::Args), /// Reachability over the compiler's call graph; coverage of what can run, and gap lists. Callgraph(tools::callgraph::Args), + /// rustc's -C and -Z options: domains, values and pairs accepted, covering-array sizes. + FlagUniverse(tools::flag_universe::Args), + /// A PICT model of the option universe (optionally of transitions, for a Cargo build). + FlagModel(tools::flag_model::Args), + /// Compile a trivial crate per PICT row; count the rejected rows by first error. + FlagRows(tools::flag_rows::Args), + /// Option transitions between incremental sessions: the rebuild must match a clean build. + FlagWalk(tools::flag_walk::Args), + /// Delta-debug the options of a flag walk's failing rows to a minimal set per error. + FlagMin(tools::flag_min::Args), + /// The fuzzer under the option configurations of a PICT table. + FlagFuzz(tools::flag_fuzz::Args), + /// [UNTRACKED] options must not change what incremental compilation reuses. + AuditOptions(tools::audit_options::Args), } fn main() -> ExitCode { @@ -82,6 +103,13 @@ fn main() -> ExitCode { Check::ScaleCheck(a) => tools::scale_check::run(a), Check::AbiDiff(a) => tools::abi_diff::run(a), Check::Callgraph(a) => tools::callgraph::run(a), + Check::FlagUniverse(a) => tools::flag_universe::run(a), + Check::FlagModel(a) => tools::flag_model::run(a), + Check::FlagRows(a) => tools::flag_rows::run(a), + Check::FlagWalk(a) => tools::flag_walk::run(a), + Check::FlagMin(a) => tools::flag_min::run(a), + Check::FlagFuzz(a) => tools::flag_fuzz::run(a), + Check::AuditOptions(a) => tools::audit_options::run(a), }; match result { Ok(code) => code, diff --git a/crates/mirth-lab/src/mutations.rs b/crates/mirth-lab/src/mutations.rs new file mode 100644 index 0000000..d7602ec --- /dev/null +++ b/crates/mirth-lab/src/mutations.rs @@ -0,0 +1,328 @@ +//! Random mechanical edits to Rust source, shared by the fuzzers and flag-walk. +//! +//! Each edit takes the file's text, a random generator and a counter, and returns the new +//! text, or None when it does not apply. EDITS lists them with weights. + +use std::sync::LazyLock; + +use fastrand::Rng; +use regex::Regex; + +pub type Edit = fn(&str, &mut Rng, usize) -> Option; + +/// Top-level items: blank-line separated, continuation blocks merged. +fn blocks(text: &str) -> Vec { + let mut out: Vec = Vec::new(); + for b in text.split("\n\n") { + let cont = b.chars().next().is_some_and(char::is_whitespace) || b.starts_with('}') || b.starts_with("where"); + match out.last_mut() { + Some(last) if cont => { + last.push_str("\n\n"); + last.push_str(b); + } + _ => out.push(b.to_owned()), + } + } + out +} + +fn indent(line: &str) -> &str { + &line[..line.len() - line.trim_start().len()] +} + +fn pick<'a, T>(rng: &mut Rng, v: &'a [T]) -> &'a T { + &v[rng.usize(..v.len())] +} + +fn comment_line(text: &str, rng: &mut Rng, n: usize) -> Option { + let mut lines: Vec = text.split('\n').map(str::to_owned).collect(); + let i = rng.usize(..=lines.len()); + let ind = lines.get(i).map_or("", |l| indent(l)).to_owned(); + lines.insert(i, format!("{ind}// fuzz {n}")); + Some(lines.join("\n")) +} + +fn blank_line(text: &str, rng: &mut Rng, _: usize) -> Option { + let mut lines: Vec<&str> = text.split('\n').collect(); + let i = rng.usize(..=lines.len()); + lines.insert(i, ""); + Some(lines.join("\n")) +} + +fn remove_comment(text: &str, rng: &mut Rng, _: usize) -> Option { + let mut lines: Vec<&str> = text.split('\n').collect(); + let idx: Vec = + (0..lines.len()).filter(|&i| lines[i].trim().starts_with("//") && !lines[i].trim().starts_with("//!")).collect(); + if idx.is_empty() { + return None; + } + lines.remove(*pick(rng, &idx)); + Some(lines.join("\n")) +} + +fn indent_line(text: &str, rng: &mut Rng, _: usize) -> Option { + let mut lines: Vec = text.split('\n').map(str::to_owned).collect(); + let idx: Vec = (0..lines.len()).filter(|&i| !lines[i].trim().is_empty()).collect(); + if idx.is_empty() { + return None; + } + let i = *pick(rng, &idx); + lines[i] = format!(" {}", lines[i]); + Some(lines.join("\n")) +} + +fn swap_items(text: &str, rng: &mut Rng, _: usize) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + let i = rng.usize(1..b.len() - 1); + b.swap(i, i + 1); + Some(b.join("\n\n")) +} + +fn move_item_to_end(text: &str, rng: &mut Rng, _: usize) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + let item = b.remove(rng.usize(1..b.len())); + b.push(item.trim_end_matches('\n').to_owned()); + Some(b.join("\n\n") + "\n") +} + +fn delete_item(text: &str, rng: &mut Rng, _: usize) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + b.remove(rng.usize(1..b.len())); + Some(b.join("\n\n")) +} + +static FN_ITEM: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^(pub(\([^)]*\))? )?(const )?(async )?fn \w+").unwrap()); +static FN_NAME: LazyLock = LazyLock::new(|| Regex::new(r"\bfn (\w+)").unwrap()); + +fn duplicate_fn(text: &str, rng: &mut Rng, n: usize) -> Option { + let mut b = blocks(text); + let fns: Vec = (0..b.len()).filter(|&i| FN_ITEM.is_match(&b[i])).collect(); + if fns.is_empty() { + return None; + } + let i = *pick(rng, &fns); + let copy = FN_NAME.replacen(&b[i], 1, |c: ®ex::Captures| format!("fn {}_fuzz{n}", &c[1])).into_owned(); + b.insert(i + 1, copy); + Some(b.join("\n\n")) +} + +/// Items to add; `{n}` is the counter. +const ADDITIONS: &[&str] = &[ + "fn fuzz_private_{n}() -> u32 { {n} }", + "pub fn fuzz_public_{n}(x: u32) -> u32 { x.wrapping_mul({n}) }", + "#[inline]\npub fn fuzz_inline_{n}(x: &T) -> (T, u32) { (x.clone(), {n}) }", + "pub const FUZZ_{n}: &str = \"fuzz {n}\";", + "pub static FUZZ_STATIC_{n}: [u8; 3] = [{n} as u8, 1, 2];", + "#[derive(Debug, Clone, PartialEq)]\npub struct Fuzz{n} { pub items: [T; N], pub tag: &'static str }", + "pub enum FuzzEnum{n} { A(u32), B { x: i64 }, C }", + "pub trait FuzzTrait{n} { fn go(&self) -> impl Sized; const K: u32 = {n}; }", + "pub async fn fuzz_async_{n}() -> u32 { {n} }", + "pub type FuzzAlias{n} = Vec<(T, u32)>;", + "macro_rules! fuzz_macro_{n} { ($e:expr) => { $e + {n} }; }", + "pub mod fuzz_mod_{n} { pub fn inner() -> &'static str { \"{n}\" } }", +]; + +fn add_item(text: &str, rng: &mut Rng, n: usize) -> Option { + let mut b = blocks(text); + let i = rng.usize(1..=b.len()); + b.insert(i, pick(rng, ADDITIONS).replace("{n}", &n.to_string())); + Some(b.join("\n\n")) +} + +fn is_word(c: Option) -> bool { + c.is_some_and(|c| c.is_alphanumeric() || c == '_') +} + +static DIGITS: LazyLock = LazyLock::new(|| Regex::new(r"\d+").unwrap()); + +fn int_literal(text: &str, rng: &mut Rng, _: usize) -> Option { + // A whole number not touching a word character or a dot. + let ms: Vec = DIGITS + .find_iter(text) + .filter(|m| { + let before = text[..m.start()].chars().next_back(); + let after = text[m.end()..].chars().next(); + !is_word(before) && before != Some('.') && !is_word(after) && after != Some('.') + }) + .collect(); + if ms.is_empty() { + return None; + } + let m = pick(rng, &ms); + let v: u128 = m.as_str().parse().ok()?; + Some(format!("{}{}{}", &text[..m.start()], v + 1, &text[m.end()..])) +} + +fn str_literal(text: &str, rng: &mut Rng, _: usize) -> Option { + // `"…"` with no quote, backslash or newline inside, not after a word character or a backslash. + let bytes = text.as_bytes(); + let mut ms: Vec<(usize, usize)> = Vec::new(); + let mut i = 0; + while i < bytes.len() { + if bytes[i] == b'"' { + let before = text[..i].chars().next_back(); + if !is_word(before) && before != Some('\\') { + let rest = &bytes[i + 1..]; + if let Some(j) = rest.iter().position(|&c| c == b'"' || c == b'\\' || c == b'\n') { + if rest[j] == b'"' { + ms.push((i, i + 1 + j + 1)); + i = i + 1 + j + 1; + continue; + } + } + } + } + i += 1; + } + if ms.is_empty() { + return None; + } + let (s, e) = *pick(rng, &ms); + Some(format!("{}\"{}~\"{}", &text[..s], &text[s + 1..e - 1], &text[e..])) +} + +static FN_LINE: LazyLock = LazyLock::new(|| Regex::new(r"^\s*(pub(\([^)]*\))? )?(const )?fn ").unwrap()); +static ITEM_LINE: LazyLock = + LazyLock::new(|| Regex::new(r"^\s*(pub(\([^)]*\))? )?(fn|struct|enum|trait|const|static|type|mod) ").unwrap()); + +fn toggle_inline(text: &str, rng: &mut Rng, _: usize) -> Option { + let mut lines: Vec = text.split('\n').map(str::to_owned).collect(); + let inl: Vec = + (0..lines.len()).filter(|&i| ["#[inline]", "#[inline(never)]", "#[inline(always)]"].contains(&lines[i].trim())).collect(); + let fns: Vec = (0..lines.len()).filter(|&i| FN_LINE.is_match(&lines[i])).collect(); + if !inl.is_empty() && rng.f64() < 0.5 { + lines.remove(*pick(rng, &inl)); + } else if !fns.is_empty() { + let i = *pick(rng, &fns); + let ind = indent(&lines[i]).to_owned(); + let attr = pick(rng, &["#[inline]", "#[inline(never)]", "#[cold]", "#[must_use]"]); + lines.insert(i, format!("{ind}{attr}")); + } else { + return None; + } + Some(lines.join("\n")) +} + +fn doc_comment(text: &str, rng: &mut Rng, n: usize) -> Option { + let mut lines: Vec = text.split('\n').map(str::to_owned).collect(); + let idx: Vec = (0..lines.len()).filter(|&i| ITEM_LINE.is_match(&lines[i])).collect(); + if idx.is_empty() { + return None; + } + let i = *pick(rng, &idx); + let ind = indent(&lines[i]).to_owned(); + lines.insert(i, format!("{ind}/// Fuzz doc {n}, see [`Vec`].")); + Some(lines.join("\n")) +} + +static DERIVE: LazyLock = LazyLock::new(|| Regex::new(r"#\[derive\(([^)]*)\)\]").unwrap()); + +fn reorder_derive(text: &str, rng: &mut Rng, _: usize) -> Option { + let ms: Vec = DERIVE.captures_iter(text).collect(); + if ms.is_empty() { + return None; + } + let m = pick(rng, &ms); + let mut names: Vec<&str> = m[1].split(',').map(str::trim).filter(|x| !x.is_empty()).collect(); + if names.len() < 2 { + return None; + } + rng.shuffle(&mut names); + let all = m.get(0).unwrap(); + Some(format!("{}#[derive({})]{}", &text[..all.start()], names.join(", "), &text[all.end()..])) +} + +static PUB_ITEM: LazyLock = LazyLock::new(|| Regex::new(r"\bpub (fn|struct|enum|const|static|trait|mod|type) ").unwrap()); + +fn narrow_visibility(text: &str, rng: &mut Rng, _: usize) -> Option { + let ms: Vec = PUB_ITEM.captures_iter(text).collect(); + if ms.is_empty() { + return None; + } + let m = pick(rng, &ms); + let all = m.get(0).unwrap(); + Some(format!("{}pub(crate) {} {}", &text[..all.start()], &m[1], &text[all.end()..])) +} + +static LET: LazyLock = LazyLock::new(|| Regex::new(r"\blet (mut )?([a-z_][a-z0-9_]*)\b").unwrap()); + +fn rename_local(text: &str, rng: &mut Rng, n: usize) -> Option { + let ms: Vec = LET.captures_iter(text).collect(); + if ms.is_empty() { + return None; + } + let m = pick(rng, &ms); + let name = &m[2]; + if name == "_" { + return None; + } + let all = m.get(0).unwrap(); + // Rename from the binding to the end of the enclosing top-level block. + let end = text[all.end()..].find("\n}\n").map_or(text.len(), |i| all.end() + i); + let word = Regex::new(&format!(r"\b{name}\b")).ok()?; + let body = word.replace_all(&text[all.start()..end], format!("{name}_f{n}").as_str()); + Some(format!("{}{}{}", &text[..all.start()], body, &text[end..])) +} + +pub const EDITS: &[(Edit, u32)] = &[ + (comment_line, 10), + (blank_line, 6), + (remove_comment, 3), + (indent_line, 4), + (swap_items, 6), + (move_item_to_end, 3), + (delete_item, 2), + (duplicate_fn, 4), + (add_item, 8), + (int_literal, 6), + (str_literal, 4), + (toggle_inline, 4), + (doc_comment, 4), + (reorder_derive, 2), + (narrow_visibility, 2), + (rename_local, 3), +]; + +/// Edits that change a literal (the fuzzers keep them out of build scripts). +pub fn is_literal_edit(e: Edit) -> bool { + std::ptr::fn_addr_eq(e, int_literal as Edit) || std::ptr::fn_addr_eq(e, str_literal as Edit) +} + +/// An edit drawn by weight. +pub fn choose(rng: &mut Rng) -> Edit { + let total: u32 = EDITS.iter().map(|e| e.1).sum(); + let mut x = rng.u32(..total); + for (e, w) in EDITS { + if x < *w { + return *e; + } + x -= w; + } + EDITS[0].0 +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn literals() { + let mut rng = Rng::with_seed(1); + assert_eq!(int_literal("x1 = 2.5 + 7;", &mut rng, 0).as_deref(), Some("x1 = 2.5 + 8;")); + assert_eq!(str_literal(r#"a"b" + "c""#, &mut rng, 0).as_deref(), Some(r#"a"b" + "c~""#)); + } + + #[test] + fn blocks_merge_continuations() { + assert_eq!(blocks("a\n\n b\n\nc").len(), 2); + } +} diff --git a/crates/mirth-lab/src/tools/audit_options.rs b/crates/mirth-lab/src/tools/audit_options.rs new file mode 100644 index 0000000..4973b4e --- /dev/null +++ b/crates/mirth-lab/src/tools/audit_options.rs @@ -0,0 +1,160 @@ +//! Audit rustc's [UNTRACKED] options for stale incremental reuse. +//! +//! An option rustc marks [UNTRACKED] is left out of the dependency-tracking hash, so changing +//! it between incremental sessions reuses the previous session's results. That is only correct +//! if the option cannot change them. For each untracked option that takes no value or a +//! boolean, this builds a crate incrementally without it, then again with it, and compares the +//! result with a clean build that has it: the .rmeta, each .rlib member (object code, with the +//! incremental session suffix removed from names), the diagnostics, and the files written. A +//! difference means the option changes output that incremental compilation reuses: it should +//! be tracked, or the reuse checked. +//! +//! With ARGs, audits those arguments instead of the options found in the checkout's +//! compiler/rustc_session/src/options.rs. The crate should have code of its own for codegen +//! options to change (non-generic functions) and a warning or two for diagnostic options to +//! change. Exits 1 when any option differs. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; + +use regex::Regex; + +use mirth_lab::artifacts::normalized_rlib; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// A rust checkout, to read the untracked options from. + #[arg(long)] + source: Option, + /// The crate root to build, as a library. + #[arg(long = "crate")] + krate: PathBuf, + #[arg(long, default_value = "audited")] + crate_name: String, + #[arg(long, default_value = "2021")] + edition: String, + #[arg(trailing_var_arg = true, allow_hyphen_values = true)] + args: Vec, +} + +// Options that stop compilation or change how arguments are read. +const SKIP: &[&str] = &["-Chelp", "-Zhelp", "-Zno-analysis", "-Zparse-crate-root-only=yes", "-Zshell-argfiles=yes"]; + +static UNTRACKED: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)^\s{4}(\w+):\s*([^=\n]+?)\s*=\s*\(([^,]*),\s*(parse_\w+),\s*\[UNTRACKED\]").unwrap()); +static CODEGEN: LazyLock = LazyLock::new(|| Regex::new(r"options! \{\s*CodegenOptions,").unwrap()); +static UNSTABLE: LazyLock = LazyLock::new(|| Regex::new(r"options! \{\s*UnstableOptions,").unwrap()); + +fn untracked_boolean_options(source: &Path) -> anyhow::Result> { + let text = std::fs::read_to_string(source.join("compiler/rustc_session/src/options.rs"))?; + let codegen = CODEGEN.find(&text).map_or(0, |m| m.start()); + let unstable = UNSTABLE.find(&text).map_or(0, |m| m.start()); + let mut found = Vec::new(); + for c in UNTRACKED.captures_iter(&text) { + let start = c.get(0).unwrap().start(); + let (name, default, parser) = (&c[1], c[3].trim(), &c[4]); + if !["parse_bool", "parse_no_value", "parse_opt_bool"].contains(&parser) || start < codegen { + continue; + } + let group = if start > unstable { "Z" } else { "C" }; + let mut flag = format!("-{group}{}", name.replace('_', "-")); + if parser != "parse_no_value" { + flag += if default == "true" || default == "Some(true)" { "=no" } else { "=yes" }; + } + if !SKIP.contains(&flag.as_str()) { + found.push(flag); + } + } + Ok(found) +} + +/// One build, always from the same working directory, which rustc records. +fn build(args: &Args, krate: &Path, work: &Path, incremental: &str, out: &str, extra: &[&str]) -> (i32, Vec) { + let _ = std::fs::create_dir_all(work.join(out)); + let r = Command::new(&args.rustc) + .args(["--edition", &args.edition, "--crate-type", "lib", "--crate-name", &args.crate_name, "--emit=metadata,link"]) + .arg(format!("-Cincremental={}", work.join(incremental).display())) + .arg("--out-dir") + .arg(work.join(out)) + .arg(krate) + .args(extra) + .current_dir(work) + .output(); + match r { + Ok(o) => { + let err = String::from_utf8_lossy(&o.stderr); + let mut diagnostics: Vec = + err.lines().filter(|l| l.starts_with("warning") || l.starts_with("error") || l.contains("-->")).map(str::to_owned).collect(); + diagnostics.sort(); + (o.status.code().unwrap_or(-1), diagnostics) + } + Err(_) => (-1, Vec::new()), + } +} + +fn listing(dir: &Path) -> BTreeSet { + std::fs::read_dir(dir).map(|r| r.flatten().map(|e| e.file_name().to_string_lossy().into_owned()).collect()).unwrap_or_default() +} + +fn audit(args: &Args, krate: &Path, flag: &str) -> Vec { + let work = tempfile::Builder::new().prefix("audit-").tempdir().expect("scratch"); + let w = work.path(); + let extra: Vec<&str> = if flag.is_empty() { vec![] } else { vec![flag] }; + let (rc0, _) = build(args, krate, w, "i", "o1", &[]); + let (rc1, diag_inc) = build(args, krate, w, "i", "o1", &extra); + let (rc2, diag_clean) = build(args, krate, w, "j", "o2", &extra); + if rc0 != 0 || rc2 != 0 { + return vec![format!("the crate does not build (exit {rc0} without the option, {rc2} with it)")]; + } + let name = &args.crate_name; + let mut problems = Vec::new(); + if rc1 != rc2 { + problems.push(format!("the incremental rebuild exits {rc1}, a clean build {rc2}")); + } + if std::fs::read(w.join(format!("o1/lib{name}.rmeta"))).ok() != std::fs::read(w.join(format!("o2/lib{name}.rmeta"))).ok() { + problems.push("metadata".into()); + } + let a = normalized_rlib(&w.join(format!("o1/lib{name}.rlib"))); + let b = normalized_rlib(&w.join(format!("o2/lib{name}.rlib"))); + let members: BTreeSet<&String> = a.keys().chain(b.keys()).filter(|m| a.get(*m) != b.get(*m)).collect(); + if !members.is_empty() { + problems.push(format!("object code ({} rlib members differ or exist on one side)", members.len())); + } + if diag_inc != diag_clean { + problems.push(format!("diagnostics ({} lines incrementally, {} clean)", diag_inc.len(), diag_clean.len())); + } + let o1 = listing(&w.join("o1")); + let missing = listing(&w.join("o2")).difference(&o1).count(); + if missing > 0 { + problems.push(format!("{missing} files only a clean build writes")); + } + problems +} + +pub fn run(args: Args) -> anyhow::Result { + let krate = std::fs::canonicalize(&args.krate)?; + let flags = if !args.args.is_empty() { + args.args.clone() + } else { + let source = args.source.as_ref().ok_or_else(|| anyhow::anyhow!("--source or ARGs needed"))?; + untracked_boolean_options(source)? + }; + let show = |p: &[String]| if p.is_empty() { "same".to_owned() } else { p.join("; ") }; + let control = audit(&args, &krate, ""); + println!("{:40} {}", "(control: no option)", show(&control)); + if !control.is_empty() { + eprintln!("the control differs: incremental and clean builds disagree without any option"); + return Ok(ExitCode::from(1)); + } + let mut any = false; + for flag in &flags { + let r = audit(&args, &krate, flag); + println!("{flag:40} {}", show(&r)); + any |= !r.is_empty(); + } + Ok(if any { ExitCode::from(1) } else { ExitCode::SUCCESS }) +} diff --git a/crates/mirth-lab/src/tools/flag_fuzz.rs b/crates/mirth-lab/src/tools/flag_fuzz.rs new file mode 100644 index 0000000..9de92b5 --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_fuzz.rs @@ -0,0 +1,92 @@ +//! Run the fuzzer under option configurations: for each chosen row of a PICT table (the B side +//! of a `flag-model --transitions --cargo` model), fuzz the fixture with those options in +//! RUSTFLAGS for a number of edits. Stops at the first finding (`mirth-lab fuzz +//! --pause-on-finding`, exit 3); rerunning resumes after the rows already done. +//! +//! Writes /row/ (the fuzzer's work directory) and /rows.jsonl (one line per +//! finished row: options, the fuzzer's totals). + +use std::collections::HashSet; +use std::io::Write as _; +use std::path::PathBuf; +use std::process::{Command, ExitCode}; + +use serde::{Deserialize, Serialize}; + +use super::flag_model; +use super::flag_walk::slice; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + fixture: PathBuf, + /// flag-universe's work directory. + #[arg(long)] + flags: PathBuf, + #[arg(long)] + table: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value = "")] + rows: String, + /// Per row, over all workers. + #[arg(long, default_value_t = 200)] + edits: usize, + #[arg(long, default_value_t = 8)] + workers: usize, +} + +#[derive(Serialize, Deserialize)] +struct Done { + row: usize, + flags: Vec, + total: String, +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let opts = flag_model::options(&args.flags)?; + let rows = flag_model::table(&args.table)?; + let log = work.join("rows.jsonl"); + let done: HashSet = std::fs::read_to_string(&log) + .unwrap_or_default() + .lines() + .filter_map(|l| serde_json::from_str::(l).ok()?["row"].as_u64().map(|r| r as usize)) + .collect(); + let me = std::env::current_exe()?; + for i in slice(&args.rows, rows.len()) { + if done.contains(&i) { + continue; + } + let flags = flag_model::row_flags(&rows[i], Some("B"), &opts); + let w = work.join(format!("row{i}")); + let out = Command::new(&me) + .arg("fuzz") + .args(["--rustc", &args.rustc, "--fixture"]) + .arg(&args.fixture) + .arg("--work") + .arg(&w) + .args(["--workers", &args.workers.to_string()]) + .args(["--edits", &(args.edits / args.workers.max(1)).max(1).to_string()]) + .args(["--seed", &i.to_string()]) + .args(["--rustflags", &flags.join(" ")]) + .args(["--target", "x86_64-unknown-linux-gnu", "--pause-on-finding"]) + .output()?; + let stdout = String::from_utf8_lossy(&out.stdout); + let total = stdout.trim().lines().last().unwrap_or("").to_owned(); + let paused = std::fs::read_to_string(w.join("PAUSED")).ok(); + match &paused { + Some(p) => println!("row {i}: {total} PAUSED {p}"), + None => println!("row {i}: {total}"), + } + if paused.is_some() { + return Ok(ExitCode::from(3)); // not recorded as done: rerun after patching to do this row again + } + let mut f = std::fs::OpenOptions::new().create(true).append(true).open(&log)?; + writeln!(f, "{}", serde_json::to_string(&Done { row: i, flags: flags[1..].to_vec(), total })?)?; + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/flag_min.rs b/crates/mirth-lab/src/tools/flag_min.rs new file mode 100644 index 0000000..6e9198c --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_min.rs @@ -0,0 +1,163 @@ +//! Minimize the failures of a flag-walk run: for each distinct first error among rows whose +//! clean build with A failed, find the smallest set of the row's options that still gives the +//! same error (delta debugging, a clean build of the fixture per test). +//! +//! Prints one line per error: the minimal options. Writes /minimized.json. + +use std::path::PathBuf; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; + +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; + +use super::flag_model::{to_json_indent1, FLAG_BASE}; +use super::flag_walk::copy_fixture; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[arg(long)] + fixture: PathBuf, + /// flag-walk's work directory. + #[arg(long)] + walk: PathBuf, + #[arg(long, default_value_t = 1)] + per_error: usize, + #[arg(long, default_value_t = 4)] + jobs: usize, + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "x86_64-unknown-linux-gnu")] + target: String, +} + +static VOLATILE: LazyLock = LazyLock::new(|| Regex::new(r"_R\w+|/\S+|`[^`]*`|\b[0-9a-f]{16}\b").unwrap()); +static DIGITS: LazyLock = LazyLock::new(|| Regex::new(r"\d+").unwrap()); + +/// An error line with paths, symbols, hashes and quoted names removed. +pub fn signature(line: &str) -> String { + let line = VOLATILE.replace_all(line, "…"); + DIGITS.replace_all(&line, "N").chars().take(120).collect() +} + +fn first_error(log: &str) -> &str { + for l in log.lines() { + if (l.starts_with("error") || l.starts_with("rustc-LLVM ERROR") || l.starts_with("LLVM ERROR")) && !l.contains("could not compile") { + return l; + } + if l.contains("panicked at") { + return l; + } + } + "" +} + +#[derive(Deserialize)] +struct Row { + #[serde(rename = "A")] + a: Vec, + #[serde(rename = "A_ok")] + a_ok: bool, + #[serde(default)] + error: Option, + #[serde(default)] + errors: Option>, +} + +#[derive(Serialize)] +struct Minimized { + error: String, + minimal: Option>, +} + +fn minimize(args: &Args, mut flags: Vec, sig: &str) -> Option> { + let work = tempfile::tempdir_in(&args.walk).expect("scratch"); + let (src, target) = (work.path().join("s"), work.path().join("t")); + copy_fixture(&args.fixture, &src).ok()?; + let fails = |fl: &[String]| -> bool { + let _ = std::fs::remove_dir_all(&target); + let rustflags = std::iter::once(FLAG_BASE.to_owned()).chain(fl.iter().cloned()).collect::>().join(" "); + let r = Command::new("cargo") + .arg(format!("+{}", args.toolchain)) + .args(["build", "--workspace", "--offline", "-j", "4", "--target", &args.target, "--target-dir"]) + .arg(&target) + .current_dir(&src) + .env("RUSTC", &args.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + .env("CARGO_TERM_COLOR", "never") + .env("RUSTFLAGS", rustflags) + .output(); + match r { + Ok(o) => !o.status.success() && signature(first_error(&String::from_utf8_lossy(&o.stderr))) == sig, + Err(_) => false, + } + }; + if !fails(&flags) { + return None; + } + let mut n = 2; + while flags.len() >= 2 { + let chunk = (flags.len() / n).max(1); + let mut reduced = false; + let mut i = 0; + while i < flags.len() { + let rest: Vec = flags[..i].iter().chain(flags[(i + chunk).min(flags.len())..].iter()).cloned().collect(); + if fails(&rest) { + flags = rest; + n = (n - 1).max(2); + reduced = true; + break; + } + i += chunk; + } + if !reduced { + if chunk == 1 { + break; + } + n = flags.len().min(n * 2); + } + } + Some(flags) +} + +pub fn run(args: Args) -> anyhow::Result { + let text = std::fs::read_to_string(args.walk.join("results.jsonl"))?; + // By signature, in first-seen order. + let mut todo: Vec<(String, Vec>)> = Vec::new(); + for line in text.lines().filter(|l| !l.trim().is_empty()) { + let r: Row = serde_json::from_str(line)?; + if r.a_ok { + continue; + } + let err = r.errors.as_ref().and_then(|e| e.first().cloned()).or(r.error.clone()).unwrap_or_default(); + let sig = signature(&err); + let i = match todo.iter().position(|(s, _)| *s == sig) { + Some(i) => i, + None => { + todo.push((sig, Vec::new())); + todo.len() - 1 + } + }; + if todo[i].1.len() < args.per_error { + todo[i].1.push(r.a); + } + } + let jobs: Vec<(&String, &Vec)> = todo.iter().flat_map(|(s, fls)| fls.iter().map(move |f| (s, f))).collect(); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let out: Vec = pool.install(|| { + jobs.par_iter().map(|(sig, fl)| Minimized { error: (*sig).clone(), minimal: minimize(&args, (*fl).clone(), sig) }).collect() + }); + std::fs::write(args.walk.join("minimized.json"), to_json_indent1(&out))?; + for o in &out { + let m = match &o.minimal { + Some(v) => format!("[{}]", v.iter().map(|s| format!("'{s}'")).collect::>().join(", ")), + None => "None".into(), + }; + println!("{m} -> {}", o.error); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/flag_model.rs b/crates/mirth-lab/src/tools/flag_model.rs new file mode 100644 index 0000000..7708089 --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_model.rs @@ -0,0 +1,342 @@ +//! A PICT model of rustc's option universe, from flag-universe's results. +//! +//! One parameter per option. Its values are absence, the values rustc accepted alone, the +//! values it accepts once `-Cunsafe-allow-abi-mismatch` names every target modifier +//! (FLAG_BASE, passed on every row), and the values that need another option, with IF/THEN +//! constraints for those needs. Options that stop compilation early (help, parse-only, +//! link-only) are left out. A value whose need lies outside the subset is dropped. +//! +//! With --cargo the model is for building a Cargo workspace (flag-walk): it leaves out the +//! values in CARGO_DROP, which fail there for reasons of Cargo or this machine, not the options. +//! +//! Combinations that hit bugs already in docs/hunt.md are excluded, so walks look for new +//! ones; --allow-known keeps those that have a local stopgap (for a compiler with them). +//! +//! With --transitions every parameter appears twice, A_ before and B_ after, for covering the +//! changes between two sessions. +//! +//! Run it with PICT (github.com/microsoft/pict): `pict /o:2` gives a pairwise covering +//! array, `/o:3` three-way. +//! +//! Also the parts the other flag tools share: options.json, table rows, a row's flags. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use regex::Regex; +use serde::{Deserialize, Serialize}; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// flag-universe's work directory. + work: PathBuf, + /// all, untracked or tracked. + subset: String, + out: PathBuf, + #[arg(long)] + transitions: bool, + #[arg(long)] + cargo: bool, + #[arg(long)] + allow_known: bool, +} + +pub const FLAG_BASE: &str = "-Cunsafe-allow-abi-mismatch=sanitizer,sanitizer-cfi-normalize-integers,\ +sanitizer-cfi-minimal-runtime,retpoline,retpoline-external-thunk,\ +indirect-branch-cs-prefix,fixed-x18,reg-struct-return,regparm,branch-protection"; + +const STOP: &[&str] = &["-Chelp", "-Zhelp", "-Zparse-crate-root-only", "-Zno-analysis", "-Zlink-only", "-Zimplicit-sysroot-deps"]; + +/// (option, values to drop; None: the whole option) +const CARGO_DROP: &[(&str, Option<&[&str]>)] = &[ + ("-Zassert-incr-state", None), // fails whenever the cache state differs, by design + ("-Zbuild-sdylib-interface", None), // Cargo's target probe fails + ("-Zchecksum-hash-algorithm", Some(&["md5", "sha1"])), // Cargo cannot parse the dep info + ("-Zdirect-access-external-data", None), // link fails + ("-Zfunction-return", Some(&["thunk-extern"])), // link fails: no thunk + ("-Zlink-native-libraries", None), // link fails + ("-Zlint-llvm-ir", None), // aborts on known LLVM lint findings (rust-lang/rust#59793) + ("-Zno-codegen", None), // later crates need the output + ("-Zno-link", None), + ("-Zpanic-in-drop", None), // std is built with unwind + ("-Zsanitizer", None), // no sanitizer runtimes in this sysroot: link fails + ("-Zretpoline-external-thunk", None), // link fails: no thunk + ("-Ztiny-const-eval-limit", None), // the fixture's const evaluation exceeds it + ("-Cpanic", Some(&["immediate-abort"])), // core is built with unwind + ("-Ccode-model", Some(&["tiny"])), // LLVM ERROR: not supported on x86_64 + ("-Ztls-model", Some(&["local-exec", "emulated"])), // dylib cannot link + // the dylib cannot link (static, pie, ropi); rwpi: finding 12 + ("-Crelocation-model", Some(&["static", "pie", "ropi", "rwpi", "ropi-rwpi"])), + ("-Clto", None), // rejected for rlibs and dylibs; Cargo's profile applies it to final artifacts only + // Makes a crate behave like the standard library, which needs stability attributes on + // `const trait`s (fixtures/sink/nightly has one). + ("-Zforce-unstable-if-unmarked", Some(&["yes"])), + // LLVM's pass listing from codegen threads interleaves with rustc's lines on stderr; a split + // -Ztime-passes-format=json line then reaches Cargo as a bare JSON message. + ("-Zprint-llvm-passes", Some(&["yes"])), + ("-Ztime-passes-format", Some(&["json"])), +]; + +/// (option, value, options needed, PICT condition) +type Need = (&'static str, &'static str, &'static [&'static str], &'static str); + +// Constraints that only a real workspace shows: a binary, a dylib, Cargo's own flags. +const CARGO_NEEDS: &[Need] = &[ + ("-Cprefer-dynamic", "yes", &["-Cpanic"], r#"[Cpanic] <> "abort""#), // libstd.so has panic_unwind + ("-Cprefer-dynamic", "yes", &["-Clto"], r#"[Clto] IN {"absent","no","off"}"#), + // Findings 13 and 14 (in LLVM, not patched): retpolines with the machine outliner, or with + // the large code model. + ("-Zretpoline", "yes", &["-Ccode-model"], r#"[Ccode_model] <> "large""#), + // Finding 16 (no stopgap): cached derive expansions with the HIR crate hash. + ("-Zcache-proc-macros", "yes", &["-Zmetadata-crate-hash"], r#"[Zmetadata_crate_hash] <> "no""#), +]; +// Bugs in docs/hunt.md that have a local stopgap: excluded unless --allow-known (for a +// compiler with the stopgaps). +const KNOWN_NEEDS: &[Need] = &[ + // Finding 9: without the default passes, local ThinLTO leaves undefined hidden symbols. + ("-Cno-prepopulate-passes", "present", &["-Zthinlto", "-Copt-level"], r#"[Zthinlto] <> "yes" AND [Copt_level] IN {"absent","0"}"#), +]; +// Values left out of every model: LLVM's machine outliner crashes in many combinations +// (finding 13), which buries everything else; and see below. +const DROP: &[(&str, &[&str])] = &[ + ("-Cllvm-args", &["-enable-machine-outliner"]), + // Not a bug: the limit counts MIR pass runs across the session, so which bodies stay under + // it depends on how many bodies the session computes (an incremental session computes + // fewer) and, with -Zthreads, on thread timing. + ("-Zmir-opt-bisect-limit", &["1", "16"]), + // Not a bug: a testing option that leaves spans out of the incremental hashes, so a rebuild + // keeps stale spans by design. + ("-Zincremental-ignore-spans", &["yes"]), +]; +// Rejected alone; accepted with FLAG_BASE or with the needs below. +const EXTRA: &[(&str, &[&str])] = &[ + ("-Zindirect-branch-cs-prefix", &["yes"]), + ("-Zretpoline-external-thunk", &["yes"]), + ("-Zretpoline", &["yes"]), + ("-Zsanitizer", &["dataflow", "memory", "safestack", "thread", "cfi", "kcfi"]), + ("-Cforce-frame-pointers", &["non-leaf"]), + ("-Cpanic", &["immediate-abort"]), + ("-Zdump-dep-graph", &["yes"]), + ("-Zsanitizer-cfi-canonical-jump-tables", &["no"]), + ("-Zsanitizer-cfi-diag", &["yes"]), + ("-Zsanitizer-cfi-generalize-pointers", &["yes"]), + ("-Zsanitizer-cfi-minimal-runtime", &["yes"]), + ("-Zsanitizer-cfi-normalize-integers", &["yes"]), + ("-Zsanitizer-cfi-recover", &["yes"]), + ("-Zsanitizer-kcfi-arity", &["yes"]), + ("-Zsplit-lto-unit", &["yes"]), + ("-Zvirtual-function-elimination", &["yes"]), +]; +const NEEDS: &[Need] = &[ + ("-Cembed-bitcode", "no", &["-Clto"], r#"[Clto] IN {"absent","no","off"}"#), + ("-Zsplit-lto-unit", "yes", &["-Clto"], r#"[Clto] IN {"yes","on","thin","fat"}"#), + ("-Zvirtual-function-elimination", "yes", &["-Clto"], r#"[Clto] IN {"yes","on","fat"}"#), + ("-Zsanitizer", "cfi", &["-Clto", "-Ccodegen-units"], r#"[Clto] IN {"yes","on","fat"} AND [Ccodegen_units] = "1""#), + ("-Zsanitizer", "kcfi", &["-Cpanic"], r#"[Cpanic] = "abort""#), + ("-Cforce-frame-pointers", "non-leaf", &["-Zunstable-options"], r#"[Zunstable_options] = "present""#), + ("-Cpanic", "immediate-abort", &["-Zunstable-options"], r#"[Zunstable_options] = "present""#), + ("-Zdump-dep-graph", "yes", &["-Zquery-dep-graph"], r#"[Zquery_dep_graph] = "yes""#), + ("-Zsanitizer-cfi-diag", "yes", &["-Zsanitizer"], r#"[Zsanitizer] = "cfi""#), + ("-Zsanitizer-cfi-recover", "yes", &["-Zsanitizer"], r#"[Zsanitizer] = "cfi""#), + ("-Zsanitizer-cfi-minimal-runtime", "yes", &["-Zsanitizer"], r#"[Zsanitizer] = "cfi""#), + ("-Zsanitizer-cfi-canonical-jump-tables", "no", &["-Zsanitizer"], r#"[Zsanitizer] = "cfi""#), + ("-Zsanitizer-cfi-generalize-pointers", "yes", &["-Zsanitizer"], r#"[Zsanitizer] IN {"cfi","kcfi"}"#), + ("-Zsanitizer-cfi-normalize-integers", "yes", &["-Zsanitizer"], r#"[Zsanitizer] IN {"cfi","kcfi"}"#), + ("-Zsanitizer-kcfi-arity", "yes", &["-Zsanitizer"], r#"[Zsanitizer] = "kcfi""#), + ( + "-Zsanitizer-cfi-minimal-runtime", + "yes", + &["-Zsanitizer-cfi-recover", "-Zsanitizer-cfi-diag"], + r#"([Zsanitizer_cfi_recover] = "yes" OR [Zsanitizer_cfi_diag] = "yes")"#, + ), +]; + +/// One -C or -Z option, as flag-universe writes it to options.json. +#[derive(Serialize, Deserialize, Clone, Debug)] +pub struct Opt { + pub flag: String, + pub name: String, + pub parser: String, + pub tracking: String, + /// None: an option without a value. + pub values: Vec>, + pub free: bool, +} + +impl Opt { + pub fn full(&self) -> String { + format!("{}{}", self.flag, self.name) + } +} + +/// One value tried alone, as flag-universe writes it to singles.json. +#[derive(Serialize, Deserialize, Clone, Debug)] +pub struct Single { + pub arg: String, + pub option: String, + pub value: Option, + pub ok: bool, + pub warn: bool, + pub msg: String, +} + +static NON_WORD: LazyLock = LazyLock::new(|| Regex::new(r"[^A-Za-z0-9]").unwrap()); +static PARAM: LazyLock = LazyLock::new(|| Regex::new(r"\[(\w+)\]").unwrap()); + +/// An option's PICT parameter name. +pub fn pname(opt: &str) -> String { + NON_WORD.replace_all(opt.trim_start_matches('-'), "_").into_owned() +} + +/// options.json of a flag-universe directory, by PICT parameter name. +pub fn options(dir: &Path) -> anyhow::Result> { + let list: Vec = serde_json::from_str(&std::fs::read_to_string(dir.join("options.json"))?)?; + Ok(list.into_iter().map(|o| (pname(&o.full()), o)).collect()) +} + +/// A PICT table: rows of (parameter, value) in column order. +pub fn table(path: &Path) -> anyhow::Result>> { + let text = std::fs::read_to_string(path)?; + let mut lines = text.lines(); + let header: Vec<&str> = lines.next().unwrap_or("").split('\t').collect(); + Ok(lines + .filter(|l| !l.is_empty()) + .map(|l| header.iter().zip(l.split('\t')).map(|(h, v)| (h.to_string(), v.to_string())).collect()) + .collect()) +} + +/// The value of `column` in a row. +pub fn get<'a>(row: &'a [(String, String)], column: &str) -> Option<&'a str> { + row.iter().find(|(k, _)| k == column).map(|(_, v)| v.as_str()) +} + +/// FLAG_BASE and a row's options; with `side` ("A" or "B"), only that side's columns. +pub fn row_flags(row: &[(String, String)], side: Option<&str>, opts: &BTreeMap) -> Vec { + let mut out = vec![FLAG_BASE.to_owned()]; + for (k, v) in row { + if v == "absent" { + continue; + } + let key = match side { + Some(s) => match k.strip_prefix(s).and_then(|r| r.strip_prefix('_')) { + Some(rest) => rest, + None => continue, + }, + None => k.as_str(), + }; + let Some(o) = opts.get(key) else { continue }; + let f = o.full(); + out.push(if v == "present" { f } else { format!("{f}={}", v.replace(';', ",")) }); + } + out +} + +/// A JSON object whose keys keep their order (serde_json's own map sorts them). +pub struct OrderedMap<'a, V>(pub &'a [(String, V)]); + +impl Serialize for OrderedMap<'_, V> { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeMap; + let mut m = s.serialize_map(Some(self.0.len()))?; + for (k, v) in self.0 { + m.serialize_entry(k, v)?; + } + m.end() + } +} + +/// JSON as Python's `json.dumps(x, indent=1)` writes it, so files stay byte-comparable. +pub fn to_json_indent1(value: &T) -> String { + let mut buf = Vec::new(); + let fmt = serde_json::ser::PrettyFormatter::with_indent(b" "); + let mut ser = serde_json::Serializer::with_formatter(&mut buf, fmt); + value.serialize(&mut ser).expect("serializable"); + let text = String::from_utf8(buf).expect("utf-8"); + // Python escapes non-ASCII. + let mut out = String::with_capacity(text.len()); + for c in text.chars() { + if c.is_ascii() { + out.push(c); + } else { + let mut units = [0u16; 2]; + for u in c.encode_utf16(&mut units) { + out.push_str(&format!("\\u{u:04x}")); + } + } + } + out +} + +pub fn run(args: Args) -> anyhow::Result { + let known = !args.allow_known; + let opts: BTreeMap = { + let list: Vec = serde_json::from_str(&std::fs::read_to_string(args.work.join("options.json"))?)?; + list.into_iter().map(|o| (o.full(), o)).collect() + }; + let singles: Vec = serde_json::from_str(&std::fs::read_to_string(args.work.join("singles.json"))?)?; + let mut domains: BTreeMap> = BTreeMap::new(); + for s in singles.iter().filter(|s| s.ok && !STOP.contains(&s.option.as_str())) { + domains.entry(s.option.clone()).or_default().push(s.value.clone().unwrap_or_else(|| "present".into())); + } + for (k, vs) in EXTRA { + let d = domains.entry(k.to_string()).or_default(); + for v in *vs { + if !d.iter().any(|x| x == v) { + d.push(v.to_string()); + } + } + } + for (k, vs) in DROP { + domains.entry(k.to_string()).or_default().retain(|v| !vs.contains(&v.as_str())); + } + if args.cargo { + for (k, vs) in CARGO_DROP { + let d = domains.entry(k.to_string()).or_default(); + match vs { + None => d.clear(), + Some(vs) => d.retain(|v| !vs.contains(&v.as_str())), + } + } + domains.retain(|_, v| !v.is_empty()); + } + let untracked = |k: &str| opts.get(k).is_some_and(|o| o.tracking == "UNTRACKED"); + let keep: Vec = + domains.keys().filter(|k| args.subset == "all" || (args.subset == "untracked") == untracked(k)).cloned().collect(); + let mut cons = Vec::new(); + let needs = NEEDS.iter().chain(if args.cargo { CARGO_NEEDS } else { &[] }).chain(if known { KNOWN_NEEDS } else { &[] }); + for (k, v, need, cond) in needs { + if !keep.iter().any(|x| x == k) { + continue; + } + if need.iter().all(|n| keep.iter().any(|x| x == n)) { + cons.push(format!("IF [{}] = \"{v}\" THEN {cond};", pname(k))); + } else if let Some(d) = domains.get_mut(*k) { + if let Some(i) = d.iter().position(|x| x == v) { + d.remove(i); + } + } + } + let mut params: Vec = keep + .iter() + .map(|k| { + let vals: Vec = std::iter::once("absent".to_owned()).chain(domains[k].iter().map(|v| v.replace(',', ";"))).collect(); + format!("{}: {}", pname(k), vals.join(", ")) + }) + .collect(); + if args.transitions { + params = ["A", "B"].iter().flat_map(|t| params.iter().map(move |p| format!("{t}_{p}"))).collect(); + cons = ["A", "B"] + .iter() + .flat_map(|t| cons.iter().map(move |c| PARAM.replace_all(c, |m: ®ex::Captures| format!("[{t}_{}]", &m[1])).into_owned())) + .collect(); + if known && keep.iter().any(|k| k == "-Zprint-type-sizes") { + // Finding 11 in docs/hunt.md: the rebuild ICEs once -Zprint-type-sizes is dropped. + cons.push(r#"IF [A_Zprint_type_sizes] = "yes" THEN [B_Zprint_type_sizes] = "yes";"#.to_owned()); + } + } + std::fs::write(&args.out, format!("{}\n\n{}\n", params.join("\n"), cons.join("\n")))?; + println!("{} parameters, {} constraints", params.len(), cons.len()); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/flag_rows.rs b/crates/mirth-lab/src/tools/flag_rows.rs new file mode 100644 index 0000000..5d2bccc --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_rows.rs @@ -0,0 +1,71 @@ +//! Compile a trivial crate once per row of a PICT table made from flag-model's model, and count +//! the rows rustc rejects, grouped by first error. +//! +//! Tables from a --transitions model are not supported. + +use std::path::PathBuf; +use std::process::{Command, ExitCode}; +use std::time::Duration; + +use rayon::prelude::*; + +use super::flag_model; +use mirth_lab::rustc::run_command; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// flag-universe's work directory (scratch directories go here too). + work: PathBuf, + table: PathBuf, + rustc: PathBuf, + /// Extra rustc arguments, e.g. --emit=metadata. + #[arg(trailing_var_arg = true, allow_hyphen_values = true)] + extra: Vec, +} + +pub fn run(args: Args) -> anyhow::Result { + let opts = flag_model::options(&args.work)?; + let rows = flag_model::table(&args.table)?; + let pool = rayon::ThreadPoolBuilder::new().num_threads(10).build()?; + let res: Vec<(bool, String, usize)> = pool.install(|| { + rows.par_iter() + .map(|row| { + let d = tempfile::tempdir_in(&args.work).expect("scratch"); + let _ = std::fs::write(d.path().join("lib.rs"), "pub fn f(x: u32) -> u32 { x.wrapping_mul(3) }\n"); + let a = flag_model::row_flags(row, None, &opts); + let mut cmd = Command::new(&args.rustc); + cmd.args(["--edition", "2021", "--crate-type", "lib"]) + .args(&args.extra) + .arg("-o") + .arg(d.path().join("out")) + .args(&a) + .arg(d.path().join("lib.rs")) + .current_dir(d.path()); + match run_command(cmd, Duration::from_secs(300)) { + Ok(f) => { + let err = f.stderr_text().lines().find(|l| l.starts_with("error")).unwrap_or("").to_owned(); + (f.success(), err, a.len() - 1) + } + Err(e) => (false, e.to_string(), a.len() - 1), + } + }) + .collect() + }); + let bad: Vec<&String> = res.iter().filter(|r| !r.0).map(|r| &r.1).collect(); + let avg = res.iter().map(|r| r.2).sum::() as f64 / res.len().max(1) as f64; + println!("{} rows, {} rejected, {avg:.0} options per row on average", rows.len(), bad.len()); + // Most common first; ties in first-seen order, as Counter.most_common. + let mut counts: Vec<(String, usize)> = Vec::new(); + for e in bad { + let k: String = e.chars().take(110).collect(); + match counts.iter_mut().find(|(x, _)| *x == k) { + Some(c) => c.1 += 1, + None => counts.push((k, 1)), + } + } + counts.sort_by(|a, b| b.1.cmp(&a.1)); + for (e, c) in counts { + println!("{c} {e}"); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/flag_universe.rs b/crates/mirth-lab/src/tools/flag_universe.rs new file mode 100644 index 0000000..542c6b4 --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_universe.rs @@ -0,0 +1,368 @@ +//! Enumerate rustc's -C and -Z options, find which values and pairs of values it accepts, and +//! size covering arrays over them. +//! +//! Each option's domain is its absence plus the values worth trying: `yes`/`no` for a boolean, +//! present for an option without a value, the values its parser's description lists for an +//! enumerated one, two samples for a number. Options taking free-form strings, paths or lists +//! are counted but left out. Every value is tried alone on a trivial crate (`--emit=metadata`, +//! so only option checking and a tiny compilation run), then every pair of accepted values of +//! different options. A pair is "rejected" when rustc fails with the pair but accepts each +//! value alone. +//! +//! Writes /options.json (domains), /singles.json, /pairs.json, and prints the +//! sizes of pairwise covering arrays built greedily over the accepted domains. + +use std::collections::{BTreeMap, HashMap, HashSet}; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::Duration; + +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; + +use super::flag_model::{to_json_indent1, Opt, OrderedMap, Single}; +use mirth_lab::rustc::{run_command, Exit}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// A rust checkout, for compiler/rustc_session/src/options.rs. + #[arg(long)] + source: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + jobs: usize, + #[arg(long)] + skip_pairs: bool, +} + +const BOOL: &[&str] = &["parse_bool", "parse_opt_bool"]; +const NO_VALUE: &[&str] = &["parse_no_value"]; +const NUMBER: &[&str] = &["parse_number", "parse_opt_number"]; +const FREE: &[&str] = &[ + "parse_string", "parse_opt_string", "parse_string_push", "parse_opt_pathbuf", "parse_list", "parse_comma_list", + "parse_opt_comma_list", "parse_ignore", "parse_target_feature", "parse_list_with_polarity", "parse_llvm_module_flag", + "parse_patchable_function_entry", "parse_autodiff", "parse_offload", "parse_allow_partial_mitigations", + "parse_deny_partial_mitigations", "parse_rust_version", "parse_unpretty", "parse_passes", "parse_branch_protection", + "parse_instrument_xray", "parse_linker_features", "parse_link_self_contained", "parse_align", "parse_location_detail", + "parse_coverage_options", "parse_codegen_retag_options", +]; + +// Values for options whose parser takes a string or whose description lists no values, picked +// by hand (`rustc --print code-models` etc. for the enumerations). +const SAMPLES: &[(&str, &[&str])] = &[ + ("-Copt-level", &["0", "1", "2", "3", "s", "z"]), + ("-Ccode-model", &["tiny", "small", "kernel", "medium", "large"]), + ("-Crelocation-model", &["static", "pic", "pie", "dynamic-no-pic", "ropi", "rwpi", "ropi-rwpi", "default"]), + ("-Ztls-model", &["global-dynamic", "local-dynamic", "initial-exec", "local-exec", "emulated"]), + ("-Ctarget-cpu", &["generic", "native", "x86-64-v2", "x86-64-v3", "x86-64-v4"]), + ("-Ctarget-feature", &["+avx2", "+avx512f", "-sse4.2", "+crt-static"]), + ("-Ztune-cpu", &["generic", "znver4"]), + ("-Zthreads", &["1", "4"]), + ("-Zlocation-detail", &["none", "file", "line,column"]), + ("-Zmin-function-alignment", &["16", "64"]), + ("-Zpatchable-function-entry", &["4", "4,2"]), + ("-Zmir-enable-passes", &["+Inline", "-GVN", "+DeadStoreElimination-final"]), + ("-Zremap-cwd-prefix", &["/remapped"]), + ("-Zsimulate-remapped-rust-src-base", &["/rustc/simulated"]), + ("-Zhint-msrv", &["1.60.0"]), + ("-Cmetadata", &["mirth"]), + ("-Zinstrument-xray", &["always", "never"]), + ("-Cllvm-args", &["-unroll-threshold=0", "-enable-machine-outliner"]), + ("-Zcrate-attr", &["allow(unused)"]), +]; + +static DESC: LazyLock = + LazyLock::new(|| Regex::new(r#"pub\(crate\) const (parse_\w+): &str =\s*((?:"(?:[^"\\]|\\.)*"\s*)+|parse_\w+|[^;]+);"#).unwrap()); +static PARSER_NAME: LazyLock = LazyLock::new(|| Regex::new(r"^parse_\w+$").unwrap()); +static BACKTICKED: LazyLock = LazyLock::new(|| Regex::new(r"`([^`]+)`").unwrap()); +static OPTION_LINE: LazyLock = LazyLock::new(|| Regex::new(r"^\s*(\w+): .*?, (parse_\w+), \[(\w+)").unwrap()); + +fn parse_options(src: &str) -> anyhow::Result> { + // The parsers' descriptions, joined across lines. + let mut descs: HashMap = DESC.captures_iter(src).map(|c| (c[1].to_owned(), c[2].to_owned())).collect(); + let aliases: Vec<(String, String)> = + descs.iter().filter(|(_, v)| PARSER_NAME.is_match(v.trim())).map(|(k, v)| (k.clone(), v.trim().to_owned())).collect(); + for (k, v) in aliases { + let target = descs.get(&v).cloned().unwrap_or_default(); + descs.insert(k, target); + } + let enum_values = |parser: &str| -> Vec { + let mut out: Vec = Vec::new(); + for c in BACKTICKED.captures_iter(descs.get(parser).map_or("", String::as_str)) { + let v = &c[1]; + if out.iter().any(|x| x == v) || v.contains(' ') || v.contains('<') || v.contains('=') { + continue; + } + out.push(v.to_owned()); + } + out + }; + let mut options = Vec::new(); + for (flag, grp) in [("-C", "CodegenOptions"), ("-Z", "UnstableOptions")] { + let i = src.find(&format!("options! {{\n {grp}")).ok_or_else(|| anyhow::anyhow!("no {grp} in options.rs"))?; + let j = i + src[i..].find("\n}").ok_or_else(|| anyhow::anyhow!("unterminated {grp}"))?; + for line in src[i..j].lines() { + let Some(m) = OPTION_LINE.captures(line) else { continue }; + let (name, parser, tracking) = (m[1].replace('_', "-"), m[2].to_owned(), m[3].to_owned()); + let mut values: Vec> = if BOOL.contains(&parser.as_str()) { + vec![Some("yes".into()), Some("no".into())] + } else if NO_VALUE.contains(&parser.as_str()) { + vec![None] + } else if NUMBER.contains(&parser.as_str()) { + vec![Some("1".into()), Some("16".into())] + } else if FREE.contains(&parser.as_str()) { + vec![] + } else { + enum_values(&parser).into_iter().map(Some).collect() + }; + let full = format!("{flag}{name}"); + if let Some((_, s)) = SAMPLES.iter().find(|(k, _)| *k == full) { + values = s.iter().map(|v| Some(v.to_string())).collect(); + } + let free = values.is_empty(); + options.push(Opt { flag: flag.into(), name, parser, tracking, values, free }); + } + } + Ok(options) +} + +fn arg(o: &Opt, v: &Option) -> String { + match v { + None => o.full(), + Some(v) => format!("{}={v}", o.full()), + } +} + +struct Tried { + ok: bool, + warn: bool, + msg: String, +} + +fn try_args(rustc: &Path, work: &Path, argv: &[&str]) -> Tried { + let d = tempfile::tempdir_in(work).expect("scratch"); + let lib = d.path().join("lib.rs"); + let _ = std::fs::write(&lib, "pub fn f(x: u32) -> u32 { x.wrapping_mul(3) }\n"); + let mut cmd = Command::new(rustc); + cmd.args(["--edition", "2021", "--crate-type", "lib", "--emit=metadata", "-o"]) + .arg(d.path().join("out.rmeta")) + .args(argv) + .arg(&lib) + .current_dir(d.path()); + match run_command(cmd, Duration::from_secs(60)) { + Ok(f) if f.exit == Exit::Timeout => Tried { ok: false, warn: false, msg: "timeout".into() }, + Ok(f) => { + let err = f.stderr_text(); + let first = err.lines().find(|l| l.starts_with("error") || l.starts_with("warning")).unwrap_or(""); + Tried { ok: f.success(), warn: err.contains("warning"), msg: first.chars().take(200).collect() } + } + Err(e) => Tried { ok: false, warn: false, msg: e.to_string() }, + } +} + +#[derive(Serialize, Deserialize)] +struct Pair { + a: String, + b: String, + ok: bool, + warn: bool, + msg: String, +} + +/// A pairwise covering array, built greedily: each row is chosen among random candidates to +/// cover the most uncovered pairs. Values are indices into each domain; index 0 is the +/// option's absence. +fn covering_array(domains: &[Vec>], forbidden: &HashSet<(Option, Option)>, seed: u64) -> Vec> { + let mut rng = fastrand::Rng::with_seed(seed); + let n = domains.len(); + let bad = |i: usize, a: usize, j: usize, b: usize| forbidden.contains(&(domains[i][a].clone(), domains[j][b].clone())); + let mut uncovered: HashSet<(usize, usize, usize, usize)> = HashSet::new(); + for i in 0..n { + for j in i + 1..n { + for a in 0..domains[i].len() { + for b in 0..domains[j].len() { + if !bad(i, a, j, b) { + uncovered.insert((i, a, j, b)); + } + } + } + } + } + let mut rows = Vec::new(); + while let Some(&target) = uncovered.iter().min() { + let (mut best, mut best_gain) = (Vec::new(), -1i64); + for _ in 0..30 { + let mut row: Vec = domains.iter().map(|d| rng.usize(..d.len())).collect(); + row[target.0] = target.1; + row[target.2] = target.3; + // repair forbidden pairs by falling back to absence + for i in 0..n { + for j in i + 1..n { + if bad(i, row[i], j, row[j]) { + if j != target.0 && j != target.2 { + row[j] = 0; + } else if i != target.0 && i != target.2 { + row[i] = 0; + } + } + } + } + let mut gain = 0i64; + for i in 0..n { + for j in i + 1..n { + gain += uncovered.contains(&(i, row[i], j, row[j])) as i64; + } + } + if gain > best_gain { + best = row; + best_gain = gain; + } + } + for i in 0..n { + for j in i + 1..n { + uncovered.remove(&(i, best[i], j, best[j])); + } + } + rows.push(best); + } + rows +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let src = std::fs::read_to_string(args.source.join("compiler/rustc_session/src/options.rs"))?; + let options = parse_options(&src)?; + std::fs::write(args.work.join("options.json"), to_json_indent1(&options))?; + let walkable: Vec<&Opt> = options.iter().filter(|o| !o.free).collect(); + println!("{} options: {} with enumerable values, {} free-form", options.len(), walkable.len(), options.len() - walkable.len()); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + + let singles_path = args.work.join("singles.json"); + let mut singles: Vec = + if singles_path.exists() { serde_json::from_str(&std::fs::read_to_string(&singles_path)?)? } else { Vec::new() }; + // Try the values not tried yet (all of them on a first run). + let done: HashSet = singles.iter().map(|s| s.arg.clone()).collect(); + let jobs: Vec<(&Opt, &Option)> = + walkable.iter().flat_map(|o| o.values.iter().map(move |v| (*o, v))).filter(|(o, v)| !done.contains(&arg(o, v))).collect(); + if !jobs.is_empty() { + let results: Vec = pool.install(|| jobs.par_iter().map(|(o, v)| try_args(&args.rustc, &args.work, &[&arg(o, v)])).collect()); + for ((o, v), r) in jobs.iter().zip(results) { + singles.push(Single { arg: arg(o, v), option: o.full(), value: (*v).clone(), ok: r.ok, warn: r.warn, msg: r.msg }); + } + std::fs::write(&singles_path, to_json_indent1(&singles))?; + } + let mut accepted: BTreeMap> = BTreeMap::new(); + for s in singles.iter().filter(|s| s.ok) { + accepted.entry(s.option.clone()).or_default().push(s.arg.clone()); + } + println!( + "{} single values tried, {} accepted ({} options with at least one)", + singles.len(), + accepted.values().map(Vec::len).sum::(), + accepted.len() + ); + + let mut rejected_pairs: HashSet<(String, String)> = HashSet::new(); + if !args.skip_pairs { + let pairs_path = args.work.join("pairs.json"); + let pairs: Vec = if pairs_path.exists() { + serde_json::from_str(&std::fs::read_to_string(&pairs_path)?)? + } else { + let opts: Vec<&String> = accepted.keys().collect(); + let mut jobs: Vec<(String, String)> = Vec::new(); + for (i, x) in opts.iter().enumerate() { + for y in &opts[i + 1..] { + for a in &accepted[*x] { + for b in &accepted[*y] { + jobs.push((a.clone(), b.clone())); + } + } + } + } + // A value rejected alone may need another option: try it with every accepted + // value of every other option. + for r in singles.iter().filter(|s| !s.ok) { + for y in opts.iter().filter(|y| ***y != r.option) { + for b in &accepted[*y] { + jobs.push((r.arg.clone(), b.clone())); + } + } + } + println!("{} pairs to try", jobs.len()); + let results: Vec = pool.install(|| jobs.par_iter().map(|(a, b)| try_args(&args.rustc, &args.work, &[a, b])).collect()); + let pairs: Vec = + jobs.into_iter().zip(results).map(|((a, b), r)| Pair { a, b, ok: r.ok, warn: r.warn, msg: r.msg }).collect(); + std::fs::write(&pairs_path, to_json_indent1(&pairs))?; + pairs + }; + let alone: HashSet<&String> = singles.iter().filter(|s| s.ok).map(|s| &s.arg).collect(); + rejected_pairs = + pairs.iter().filter(|x| !x.ok && alone.contains(&x.a) && alone.contains(&x.b)).map(|x| (x.a.clone(), x.b.clone())).collect(); + // Insertion-ordered, as the Python dict was. + let mut requires: Vec<(String, Vec)> = Vec::new(); + for x in pairs.iter().filter(|x| x.ok && !alone.contains(&x.a)) { + match requires.iter_mut().find(|(k, _)| *k == x.a) { + Some((_, v)) => v.push(x.b.clone()), + None => requires.push((x.a.clone(), vec![x.b.clone()])), + } + } + println!( + "{} pairs tried; {} pairs of values accepted alone are rejected together; {} values rejected alone are accepted with another option", + pairs.len(), + rejected_pairs.len(), + requires.len() + ); + std::fs::write(args.work.join("requires.json"), to_json_indent1(&OrderedMap(&requires)))?; + } + + let forbidden: HashSet<(Option, Option)> = rejected_pairs + .iter() + .flat_map(|(a, b)| [(Some(a.clone()), Some(b.clone())), (Some(b.clone()), Some(a.clone()))]) + .collect(); + let byopt: HashMap = options.iter().map(|o| (o.full(), o)).collect(); + let mut results: Vec<(String, usize)> = Vec::new(); + for (label, pred) in [ + ("untracked", (|o: &Opt| o.tracking == "UNTRACKED") as fn(&Opt) -> bool), + ("tracked", |o: &Opt| o.tracking != "UNTRACKED"), + ("all", |_: &Opt| true), + ] { + let opts: Vec<&String> = accepted.keys().filter(|k| byopt.get(*k).is_some_and(|o| pred(o))).collect(); + if opts.len() <= 1 { + continue; + } + let domains: Vec>> = + opts.iter().map(|o| std::iter::once(None).chain(accepted[*o].iter().cloned().map(Some)).collect()).collect(); + let mut sizes: Vec = domains.iter().map(Vec::len).collect(); + sizes.sort_by(|a, b| b.cmp(a)); + let total: f64 = sizes.iter().map(|&s| s as f64).product(); + let lower = if sizes.len() > 1 { sizes[0] * sizes[1] } else { sizes[0] }; + let rows = covering_array(&domains, &forbidden, 0); + println!( + "{label}: {} options, all combinations {}, pairwise covering array {} rows (lower bound {lower})", + opts.len(), + sci(total), + rows.len() + ); + results.push((label.to_owned(), rows.len())); + } + let covering: Vec = results.iter().map(|(k, v)| format!("\"{k}\": {v}")).collect(); + std::fs::write(args.work.join("covering.json"), format!("{{{}}}", covering.join(", ")))?; + Ok(ExitCode::SUCCESS) +} + +/// Python's `{:.3e}`. +fn sci(x: f64) -> String { + let s = format!("{x:.3e}"); + match s.split_once('e') { + Some((m, e)) => { + let e: i32 = e.parse().unwrap_or(0); + format!("{m}e{}{:02}", if e < 0 { '-' } else { '+' }, e.abs()) + } + None => s, + } +} diff --git a/crates/mirth-lab/src/tools/flag_walk.rs b/crates/mirth-lab/src/tools/flag_walk.rs new file mode 100644 index 0000000..2fd8898 --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_walk.rs @@ -0,0 +1,515 @@ +//! Walk option transitions on a fixture: for each row of a PICT table made from a +//! `flag-model --transitions` model, build the fixture clean with the A_ options, rebuild it +//! incrementally with the B_ options, build it clean with the B_ options, and compare the +//! rebuild with the clean build (metadata, object code, binary, diagnostics, the binary's +//! output), as the fuzzer does after an edit. +//! +//! Rows change options only, unless --edits asks for random source edits between A and B. +//! +//! The options go in RUSTFLAGS with `--target` set, so they apply to the fixture's crates but +//! not to its build scripts and proc macros. RUSTC_VERIFY_REUSE and RUSTC_REPORT_UNTRACKED are +//! set, for a compiler with mirth's local patches. +//! +//! Writes /results.jsonl (one line per row) and /findings// for each row whose +//! rebuild differs from the clean build, or which crashed. +//! +//! To stay at the frontier: with --pause-on-finding the walk stops taking rows at the first +//! finding not marked known and writes /PAUSED. Patch the compiler, then run the same +//! command with --rustc --recheck: the rows with findings run again first, then the +//! rows not yet walked. A rerun never repeats rows that are done. + +use std::collections::BTreeMap; +use std::io::Write as _; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Mutex; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; +use serde_json::Value; + +use super::flag_model::{self, to_json_indent1, Opt}; +use mirth_lab::artifacts::{self, py_repr, Collected}; +use mirth_lab::mutations; +use mirth_lab::rustc::{run_command, Exit}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + fixture: PathBuf, + /// flag-universe's work directory. + #[arg(long)] + flags: PathBuf, + #[arg(long)] + table: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + workers: usize, + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "x86_64-unknown-linux-gnu")] + target: String, + #[arg(long, default_value_t = 600)] + timeout: u64, + /// a:b, a slice of the table. + #[arg(long, default_value = "")] + rows: String, + /// Random source edits between A and B (the fuzzer's). + #[arg(long, default_value_t = 0)] + edits: usize, + #[arg(long, default_value_t = 0)] + seed: u64, + /// Stop taking rows at the first finding not marked known; rerun to resume. + #[arg(long)] + pause_on_finding: bool, + /// On resume, run the rows that had findings again first (after patching rustc). + #[arg(long)] + recheck: bool, + /// Clean rebuilds before a difference counts as reuse. + #[arg(long, default_value_t = 12)] + p5_builds: usize, +} + +/// Copy a fixture, leaving out build output and edit scratch at any depth. +pub fn copy_fixture(from: &Path, to: &Path) -> std::io::Result<()> { + let skip = |n: &std::ffi::OsStr| n == "target" || n == "edits" || n == "edit"; + for e in walkdir::WalkDir::new(from).follow_links(true).into_iter().filter_entry(|e| e.depth() == 0 || !skip(e.file_name())) { + let e = e?; + let dest = to.join(e.path().strip_prefix(from).expect("under the fixture")); + if e.file_type().is_dir() { + std::fs::create_dir_all(&dest)?; + } else { + std::fs::copy(e.path(), &dest)?; + } + } + Ok(()) +} + +/// Python's slice `a:b` of 0..n. +pub fn slice(spec: &str, n: usize) -> Vec { + if spec.is_empty() { + return (0..n).collect(); + } + let (lo, hi) = spec.split_once(':').unwrap_or((spec, "")); + let at = |s: &str, default: usize| -> usize { + match s.parse::() { + Ok(v) if v < 0 => (n as i64 + v).max(0) as usize, + Ok(v) => (v as usize).min(n), + Err(_) => default, + } + }; + (at(lo, 0)..at(hi, n)).collect() +} + +/// Python's repr of a JSON value (lists of strings and numbers, None). +pub fn py_value(v: &Value) -> String { + match v { + Value::Null => "None".into(), + Value::Bool(b) => if *b { "True" } else { "False" }.into(), + Value::String(s) => py_repr(s), + Value::Array(a) => format!("[{}]", a.iter().map(py_value).collect::>().join(", ")), + other => other.to_string(), + } +} + +fn py_list(v: &[String]) -> String { + format!("[{}]", v.iter().map(|s| py_repr(s)).collect::>().join(", ")) +} + +#[derive(Serialize, Deserialize, Clone, Debug)] +pub struct RowResult { + pub row: usize, + #[serde(rename = "A")] + pub a: Vec, + #[serde(rename = "B")] + pub b: Vec, + #[serde(rename = "A_ok")] + pub a_ok: bool, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub diff: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub inc_ok: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub clean_ok: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub reuse: Option>, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub untracked: Option>, + #[serde(default)] + pub error: String, + #[serde(default)] + pub errors: Vec, + #[serde(default)] + pub findings: Vec, + #[serde(default)] + pub rustc: String, +} + +struct Build { + ok: bool, + log: String, + ice: bool, + reuse: Vec, + untracked: Vec, + error: String, + errors: Vec, + art: Option, + exe: Option, +} + +struct Walk<'a> { + args: &'a Args, + opts: BTreeMap, + fixture: PathBuf, + work: PathBuf, + results: Mutex<()>, +} + +fn tail(s: &str, n: usize) -> &str { + let mut i = s.len().saturating_sub(n); + while !s.is_char_boundary(i) { + i += 1; + } + &s[i..] +} + +impl Walk<'_> { + /// RUSTFLAGS for one side of a row. Cargo passes `-Cembed-bitcode=no` unless its profile + /// asks for LTO, and the profile's LTO reaches only the final artifacts, so `-Clto` goes in + /// RUSTFLAGS with `-Cembed-bitcode=yes` after Cargo's flag. + fn flags(&self, row: &[(String, String)], side: &str) -> Vec { + let mut out = flag_model::row_flags(row, Some(side), &self.opts); + let lto = flag_model::get(row, &format!("{side}_Clto")); + if matches!(lto, Some("yes" | "on" | "thin" | "fat")) && flag_model::get(row, &format!("{side}_Cembed_bitcode")) == Some("absent") { + out.push("-Cembed-bitcode=yes".into()); + } + out + } + + fn build(&self, src: &Path, target: &Path, rustflags: &[String]) -> Build { + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{}", self.args.toolchain)) + .args(["build", "--workspace", "--offline", "-j", "4", "--target", &self.args.target, "--target-dir"]) + .arg(target) + .arg("--message-format=json-render-diagnostics") + .current_dir(src) + .env("RUSTC", &self.args.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + .env("CARGO_TERM_COLOR", "never") + .env("RUSTFLAGS", rustflags.join(" ")) + .env("RUSTC_VERIFY_REUSE", "1") + .env("RUSTC_REPORT_UNTRACKED", "1"); + let (ok, out, log) = match run_command(cmd, Duration::from_secs(self.args.timeout)) { + Ok(f) => { + let mut log = f.stderr_text(); + if f.exit == Exit::Timeout { + log.push_str(&format!("\nkilled after {}s", self.args.timeout)); + } + (f.success(), f.stdout_text(), log) + } + Err(e) => (false, String::new(), e.to_string()), + }; + let name = self.fixture.file_name().map(|n| n.to_string_lossy().into_owned()).unwrap_or_default(); + let mut exe = None; + for line in out.lines() { + let Ok(msg) = serde_json::from_str::(line) else { continue }; + if msg["reason"] == "compiler-artifact" && msg["target"]["name"] == name.as_str() { + if let Some(e) = msg["executable"].as_str().filter(|e| !e.is_empty()) { + exe = Some(e.to_owned()); + } + } + } + let mut reuse: Vec = log + .lines() + .filter(|l| l.starts_with("rustc-verify-reuse:")) + .map(|l| l.split_once(':').unwrap().1.trim().chars().take(200).collect()) + .collect(); + reuse.sort(); + reuse.dedup(); + let mut untracked: Vec = log.lines().filter(|l| l.starts_with("rustc-untracked-read:")).map(|l| l.trim().to_owned()).collect(); + untracked.sort(); + untracked.dedup(); + Build { + ok, + ice: log.contains("internal compiler error") || log.contains("the compiler unexpectedly panicked") || log.contains("rustc interrupted by SIG"), + reuse, + untracked, + error: log.lines().find(|l| l.starts_with("error")).unwrap_or("").to_owned(), + errors: log + .lines() + .filter(|l| (l.starts_with("error") || l.starts_with("rustc-LLVM ERROR") || l.starts_with("LLVM ERROR")) && !l.contains("could not compile")) + .take(6) + .map(|l| l.chars().take(300).collect()) + .collect(), + art: ok.then(|| artifacts::collect(&out, target)), + exe, + log, + } + } + + /// Apply `n` random edits to the fixture's sources; returns the unified diff. + fn edit(&self, src: &Path, rng: &mut fastrand::Rng, n: usize) -> String { + let mut diff = String::new(); + for k in 0..n { + let mut paths: Vec = walkdir::WalkDir::new(src) + .into_iter() + .filter_map(Result::ok) + .map(|e| e.into_path()) + .filter(|p| p.extension().is_some_and(|x| x == "rs") && !p.strip_prefix(src).unwrap().components().any(|c| c.as_os_str() == "target")) + .collect(); + paths.sort(); + if paths.is_empty() { + break; + } + for _ in 0..20 { + let path = &paths[rng.usize(..paths.len())]; + let f = mutations::choose(rng); + if mutations::is_literal_edit(f) && path.file_name().is_some_and(|n| n == "build.rs") { + continue; // stale OUT_DIR files, or a build script that loops + } + let Ok(old) = std::fs::read_to_string(path) else { continue }; + match f(&old, rng, k) { + Some(new) if new != old => { + let _ = std::fs::write(path, &new); + let rel = path.strip_prefix(src).unwrap().display().to_string(); + let d = similar::TextDiff::from_lines(&old, &new); + diff += &d.unified_diff().header(&format!("a/{rel}"), &format!("b/{rel}")).to_string(); + break; + } + _ => {} + } + } + } + diff + } + + fn run_exe(exe: Option<&str>) -> Value { + let Some(exe) = exe else { return Value::Null }; + match run_command(Command::new(exe), Duration::from_secs(30)) { + Ok(f) if f.exit == Exit::Timeout => serde_json::json!(["timeout", ""]), + Ok(f) => { + let code = match f.exit { + Exit::Code(c) => c, + Exit::Signal(s) => -s, + Exit::Timeout => unreachable!(), + }; + serde_json::json!([code, tail(&f.stdout_text(), 2000)]) + } + Err(_) => Value::Null, + } + } + + fn write_finding(&self, i: usize, res: &RowResult, findings: &[String], logs: &[(&str, &str)]) { + let d = self.work.join("findings").join(format!("r{i}")); + let _ = std::fs::create_dir_all(&d); + let mut r = res.clone(); + r.findings = findings.to_vec(); + let _ = std::fs::write(d.join("row.json"), to_json_indent1(&r)); + if let Some(diff) = res.diff.as_ref().filter(|d| !d.is_empty()) { + let _ = std::fs::write(d.join("edit.diff"), diff); + } + for (name, log) in logs { + let _ = std::fs::write(d.join(name), tail(log, 20000)); + } + } + + fn walk(&self, i: usize, row: &[(String, String)]) { + if self.work.join("PAUSED").exists() || self.work.join("STOP").exists() { + return; + } + let home = self.work.join(format!("r{i}")); + let _ = std::fs::remove_dir_all(&home); + let (src, target, inc_target) = (home.join("src"), home.join("target"), home.join("target-inc")); + if let Err(e) = copy_fixture(&self.fixture, &src) { + eprintln!("row {i}: cannot copy the fixture: {e}"); + return; + } + let (a, b) = (self.flags(row, "A"), self.flags(row, "B")); + let mut res = RowResult { + row: i, + a: a[1..].to_vec(), + b: b[1..].to_vec(), + a_ok: false, + diff: None, + inc_ok: None, + clean_ok: None, + reuse: None, + untracked: None, + error: String::new(), + errors: vec![], + findings: vec![], + rustc: String::new(), + }; + let first = self.build(&src, &target, &a); + res.a_ok = first.ok; + let mut findings: Vec = Vec::new(); + if first.ice { + findings.push("ICE in clean A".into()); + } + if first.ok { + let mut rng = fastrand::Rng::with_seed(self.args.seed.wrapping_mul(1_000_003).wrapping_add(i as u64)); + res.diff = Some(self.edit(&src, &mut rng, self.args.edits)); + let inc = self.build(&src, &target, &b); + let _ = std::fs::rename(&target, &inc_target); + let clean = self.build(&src, &target, &b); + res.inc_ok = Some(inc.ok); + res.clean_ok = Some(clean.ok); + res.reuse = Some(inc.reuse.clone()); + res.untracked = Some(inc.untracked.clone()); + if inc.ice { + findings.push("ICE in rebuild B".into()); + } + if clean.ice { + findings.push("ICE in clean B".into()); + } + if inc.ok != clean.ok { + findings.push(format!( + "split: rebuild {}, clean {}", + if inc.ok { "ok" } else { "failed" }, + if clean.ok { "ok" } else { "failed" } + )); + } + if inc.ok && clean.ok { + let (ia, ca) = (inc.art.as_ref().unwrap(), clean.art.as_ref().unwrap()); + let mut diff = artifacts::compare(ia, ca); + if matches!(flag_model::get(row, "B_Csplit_debuginfo"), Some("packed" | "unpacked")) { + // Objects and binary name .dwo files by session (DW_AT_GNU_dwo_name, and + // the dwo_id hashed from it), so two clean builds differ too. + diff.remove("rlib"); + diff.remove("exe"); + } + let inc_exe = inc.exe.as_ref().map(|e| e.replace(&*target.to_string_lossy(), &inc_target.to_string_lossy())); + let ra = Self::run_exe(inc_exe.as_deref()); + let rb = Self::run_exe(clean.exe.as_deref()); + if ra != rb { + findings.push(format!("run: {} vs {}", py_value(&ra), py_value(&rb))); + } + if !diff.is_empty() { + // Clean builds may differ among themselves (P5), sometimes only one time in + // five: build clean again up to --p5-builds times before calling it reuse. + let mut p5 = BTreeMap::new(); + for _ in 0..self.args.p5_builds { + let _ = std::fs::remove_dir_all(&target); + let again = self.build(&src, &target, &b); + if let Some(art) = &again.art { + p5 = artifacts::compare(ca, art); + } + if !p5.is_empty() || !again.ok { + break; + } + } + let kind = if p5.is_empty() { "" } else { "P5 " }; + for (k, v) in &diff { + findings.push(format!("{kind}{k}: {}", py_list(&v[..v.len().min(5)]))); + } + } + } + res.error = if inc.error.is_empty() { clean.error.clone() } else { inc.error.clone() }; + res.errors = if inc.errors.is_empty() { clean.errors.clone() } else { inc.errors.clone() }; + if !findings.is_empty() { + self.write_finding(i, &res, &findings, &[("inc.log", &inc.log), ("clean.log", &clean.log)]); + } + } else { + res.error = first.error.clone(); + res.errors = first.errors.clone(); + if !findings.is_empty() { + self.write_finding(i, &res, &findings, &[("a.log", &first.log)]); + } + } + res.findings = findings.clone(); + res.rustc = self.args.rustc.clone(); + let new: Vec<&String> = findings.iter().filter(|f| !f.starts_with("known")).collect(); + if !new.is_empty() && self.args.pause_on_finding { + #[derive(Serialize)] + struct Paused<'a> { + row: usize, + findings: Vec<&'a String>, + } + let _ = std::fs::write(self.work.join("PAUSED"), to_json_indent1(&Paused { row: i, findings: new })); + } + let _ = std::fs::remove_dir_all(&home); + { + let _lock = self.results.lock().unwrap(); + if let Ok(mut f) = std::fs::OpenOptions::new().create(true).append(true).open(self.work.join("results.jsonl")) { + let _ = writeln!(f, "{}", serde_json::to_string(&res).unwrap()); + } + } + let mut line = format!("row {i}: A {}", if res.a_ok { "ok" } else { "failed" }); + if res.a_ok { + line += &format!(", rebuild {}", if res.inc_ok == Some(true) { "ok" } else { "failed" }); + } + if !findings.is_empty() { + line += &format!("; {}", py_list(&findings)); + } + if !res.error.is_empty() { + line += &format!("; {}", res.error.chars().take(100).collect::()); + } + println!("{line}"); + } +} + +/// The last result of each row, from /results.jsonl. +fn latest(work: &Path) -> BTreeMap { + let mut out = BTreeMap::new(); + for line in std::fs::read_to_string(work.join("results.jsonl")).unwrap_or_default().lines() { + if let Ok(r) = serde_json::from_str::(line) { + out.insert(r.row, r); + } + } + out +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let fixture = std::fs::canonicalize(&args.fixture)?; + let _ = std::fs::remove_file(work.join("PAUSED")); + let rows = flag_model::table(&args.table)?; + let idx = slice(&args.rows, rows.len()); + // Resume: rows with a result are done, except, with --recheck, those with findings, which + // run first. + let done = latest(&work); + let again: Vec = + if args.recheck { idx.iter().copied().filter(|i| done.get(i).is_some_and(|r| !r.findings.is_empty())).collect() } else { vec![] }; + let mut todo = again.clone(); + todo.extend(idx.iter().copied().filter(|i| !done.contains_key(i))); + if !again.is_empty() { + println!("rechecking rows {:?}", again); + } + let walk = Walk { opts: flag_model::options(&args.flags)?, fixture, work: work.clone(), results: Mutex::new(()), args: &args }; + // Rows in table order, a worker taking the next one when it is free. + let next = AtomicUsize::new(0); + std::thread::scope(|s| { + for _ in 0..args.workers.max(1) { + s.spawn(|| { + loop { + let k = next.fetch_add(1, Ordering::SeqCst); + let Some(&i) = todo.get(k) else { break }; + walk.walk(i, &rows[i]); + } + }); + } + }); + let results: Vec = latest(&work).into_values().collect(); + let mut summary = format!( + "{{\"rows\": {}, \"done\": {}, \"A ok\": {}, \"compared\": {}, \"findings\": {}", + rows.len(), + results.len(), + results.iter().filter(|r| r.a_ok).count(), + results.iter().filter(|r| r.inc_ok == Some(true) && r.clean_ok == Some(true)).count(), + results.iter().filter(|r| !r.findings.is_empty()).count() + ); + if let Ok(p) = std::fs::read_to_string(work.join("PAUSED")) { + if let Ok(v) = serde_json::from_str::(&p) { + let f: Vec = v["findings"].as_array().into_iter().flatten().map(|x| x.to_string()).collect(); + summary += &format!(", \"paused\": {{\"row\": {}, \"findings\": [{}]}}", v["row"], f.join(", ")); + } + } + println!("{summary}}}"); + Ok(ExitCode::SUCCESS) +} From cb6d4d43937b815938605fea4ab2d9c67577a5c1 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:02:36 +0000 Subject: [PATCH 16/23] mirth-lab: coverage, coverage-compact, coverage-generators, grammar-coverage, ui-coverage (validated against the Python scripts); shell callers build and use mirth-lab Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- crates/mirth-lab/src/coverage.rs | 170 ++++++++++ crates/mirth-lab/src/lib.rs | 1 + crates/mirth-lab/src/main.rs | 20 ++ crates/mirth-lab/src/tools/coverage.rs | 111 +++++++ .../mirth-lab/src/tools/coverage_compact.rs | 100 ++++++ .../src/tools/coverage_generators.rs | 309 ++++++++++++++++++ .../mirth-lab/src/tools/grammar_coverage.rs | 268 +++++++++++++++ crates/mirth-lab/src/tools/ui_coverage.rs | 270 +++++++++++++++ rustc/coverage-fulldeps.sh | 4 +- rustc/coverage-report.sh | 1 + rustc/coverage-run.sh | 6 +- rustc/coverage-suites.sh | 8 +- 12 files changed, 1262 insertions(+), 6 deletions(-) create mode 100644 crates/mirth-lab/src/coverage.rs create mode 100644 crates/mirth-lab/src/tools/coverage.rs create mode 100644 crates/mirth-lab/src/tools/coverage_compact.rs create mode 100644 crates/mirth-lab/src/tools/coverage_generators.rs create mode 100644 crates/mirth-lab/src/tools/grammar_coverage.rs create mode 100644 crates/mirth-lab/src/tools/ui_coverage.rs diff --git a/crates/mirth-lab/src/coverage.rs b/crates/mirth-lab/src/coverage.rs new file mode 100644 index 0000000..b724668 --- /dev/null +++ b/crates/mirth-lab/src/coverage.rs @@ -0,0 +1,170 @@ +//! What a coverage-instrumented compiler (rustc/coverage.toml) writes: the site tables of its +//! build (`.sites`) and the `V ` lines each process run with MIRTH_OUT logs. + +use std::collections::BTreeSet; +use std::io::{self, Write}; +use std::path::{Path, PathBuf}; + +use serde_json::ser::Formatter; + +/// An instrumented function: a `cover` site. +pub struct Function { + pub site: String, + pub krate: String, + pub path: String, + pub span: String, +} + +/// The files of `dir` with extension `ext`, not recursing. +pub fn files_with(dir: &Path, ext: &str) -> Vec { + let Ok(entries) = std::fs::read_dir(dir) else { return Vec::new() }; + entries.flatten().map(|e| e.path()).filter(|p| p.extension().is_some_and(|x| x == ext)).collect() +} + +fn read_lossy(path: &Path) -> String { + String::from_utf8_lossy(&std::fs::read(path).unwrap_or_default()).into_owned() +} + +/// Every `cover` site of the tables in `dir`. +pub fn functions(dir: &Path) -> Vec { + let mut out = Vec::new(); + for table in files_with(dir, "sites") { + for line in read_lossy(&table).lines() { + let f: Vec<&str> = line.split('\t').collect(); + if f.len() >= 7 && f[1] == "cover" { + out.push(Function { site: f[0].into(), krate: f[3].into(), path: f[4].into(), span: f[6].into() }); + } + } + } + out +} + +/// The sites a log reached. +pub fn log_hits(text: &str) -> impl Iterator { + text.lines().filter_map(|l| l.strip_prefix("V\t")) +} + +/// The sites the logs in `dir` reached, and how many logs there were. +pub fn hits(dir: &Path) -> (BTreeSet, usize) { + let logs = files_with(dir, "log"); + let mut out = BTreeSet::new(); + for log in &logs { + out.extend(log_hits(&read_lossy(log)).map(str::to_owned)); + } + (out, logs.len()) +} + +/// Python's json.dumps spacing (`, ` and `: `) and escaping (non-ASCII as `\\uXXXX`), on one +/// line or, with `indent`, as `json.dumps(…, indent=n)` writes it. +struct Python<'a> { + indent: Option<&'a [u8]>, + depth: usize, + has_value: bool, +} + +impl Python<'_> { + fn newline(&self, w: &mut W) -> io::Result<()> { + if let Some(indent) = self.indent { + w.write_all(b"\n")?; + for _ in 0..self.depth { + w.write_all(indent)?; + } + } + Ok(()) + } + fn open(&mut self, w: &mut W, c: &[u8]) -> io::Result<()> { + self.depth += 1; + self.has_value = false; + w.write_all(c) + } + fn close(&mut self, w: &mut W, c: &[u8]) -> io::Result<()> { + self.depth -= 1; + if self.has_value { + self.newline(w)?; + } + self.has_value = true; + w.write_all(c) + } + fn item(&mut self, w: &mut W, first: bool) -> io::Result<()> { + match (first, self.indent.is_some()) { + (true, _) => {} + (false, true) => w.write_all(b",")?, + (false, false) => w.write_all(b", ")?, + } + self.newline(w) + } +} + +impl Formatter for Python<'_> { + fn begin_array(&mut self, w: &mut W) -> io::Result<()> { + self.open(w, b"[") + } + fn end_array(&mut self, w: &mut W) -> io::Result<()> { + self.close(w, b"]") + } + fn begin_array_value(&mut self, w: &mut W, first: bool) -> io::Result<()> { + self.item(w, first) + } + fn end_array_value(&mut self, _: &mut W) -> io::Result<()> { + self.has_value = true; + Ok(()) + } + fn begin_object(&mut self, w: &mut W) -> io::Result<()> { + self.open(w, b"{") + } + fn end_object(&mut self, w: &mut W) -> io::Result<()> { + self.close(w, b"}") + } + fn begin_object_key(&mut self, w: &mut W, first: bool) -> io::Result<()> { + self.item(w, first) + } + fn begin_object_value(&mut self, w: &mut W) -> io::Result<()> { + w.write_all(b": ") + } + fn end_object_value(&mut self, _: &mut W) -> io::Result<()> { + self.has_value = true; + Ok(()) + } + fn write_string_fragment(&mut self, w: &mut W, fragment: &str) -> io::Result<()> { + for c in fragment.chars() { + if c.is_ascii() { + w.write_all(&[c as u8])?; + } else { + for unit in c.encode_utf16(&mut [0; 2]) { + write!(w, "\\u{unit:04x}")?; + } + } + } + Ok(()) + } +} + +fn python_json(value: &impl serde::Serialize, indent: Option<&[u8]>) -> String { + let mut buf = Vec::new(); + let mut ser = serde_json::Serializer::with_formatter(&mut buf, Python { indent, depth: 0, has_value: false }); + value.serialize(&mut ser).expect("serializable"); + String::from_utf8(buf).expect("utf-8") +} + +/// One line of JSON as Python's `json.dumps(value)` writes it. +pub fn to_json_line(value: &impl serde::Serialize) -> String { + python_json(value, None) +} + +/// JSON as Python's `json.dumps(value, indent=n)` writes it. +pub fn to_json_indent(value: &impl serde::Serialize, n: usize) -> String { + python_json(value, Some(&b" "[..n])) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn python_spacing() { + let v = serde_json::json!({"a": [1, 2], "b": {}, "c": "é", "d": []}); + assert_eq!(to_json_line(&v), r#"{"a": [1, 2], "b": {}, "c": "\u00e9", "d": []}"#); + assert_eq!(to_json_indent(&v, 1), "{\n \"a\": [\n 1,\n 2\n ],\n \"b\": {},\n \"c\": \"\\u00e9\",\n \"d\": []\n}"); + assert_eq!(to_json_indent(&serde_json::json!({"t": [0]}), 0), "{\n\"t\": [\n0\n]\n}"); + } +} diff --git a/crates/mirth-lab/src/lib.rs b/crates/mirth-lab/src/lib.rs index 0d8d916..84f871f 100644 --- a/crates/mirth-lab/src/lib.rs +++ b/crates/mirth-lab/src/lib.rs @@ -2,6 +2,7 @@ //! `mirth-lab` binary, in `src/tools/`; docs/checks.md says what each looks for and found. pub mod artifacts; +pub mod coverage; pub mod driver; pub mod miri; pub mod normalize; diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs index c4dbcea..e26fc37 100644 --- a/crates/mirth-lab/src/main.rs +++ b/crates/mirth-lab/src/main.rs @@ -8,9 +8,13 @@ use clap::{Parser, Subcommand}; mod tools { pub mod abi_diff; pub mod callgraph; + pub mod coverage; + pub mod coverage_compact; + pub mod coverage_generators; pub mod crash_diff; pub mod diag_check; pub mod gate_check; + pub mod grammar_coverage; pub mod instr_check; pub mod miri_diff; pub mod opt_diff; @@ -20,6 +24,7 @@ mod tools { pub mod scale_check; pub mod solver_diff; pub mod suggest_diff; + pub mod ui_coverage; pub mod xlink; } @@ -62,6 +67,16 @@ enum Check { AbiDiff(tools::abi_diff::Args), /// Reachability over the compiler's call graph; coverage of what can run, and gap lists. Callgraph(tools::callgraph::Args), + /// Which of the compiler's functions ran, per crate and file, from coverage logs. + Coverage(tools::coverage::Args), + /// Fold coverage logs into a running union as they finish, and delete them. + CoverageCompact(tools::coverage_compact::Args), + /// Compiler runs the test suites hardly make (prints, every target, links, dumps), for coverage. + CoverageGenerators(tools::coverage_generators::Args), + /// Which alternatives and tokens of Ur's Rust grammar a fixture's sources use. + GrammarCoverage(tools::grammar_coverage::Args), + /// The functions each UI test reaches beyond a baseline; a small set reaching the most. + UiCoverage(tools::ui_coverage::Args), } fn main() -> ExitCode { @@ -82,6 +97,11 @@ fn main() -> ExitCode { Check::ScaleCheck(a) => tools::scale_check::run(a), Check::AbiDiff(a) => tools::abi_diff::run(a), Check::Callgraph(a) => tools::callgraph::run(a), + Check::Coverage(a) => tools::coverage::run(a), + Check::CoverageCompact(a) => tools::coverage_compact::run(a), + Check::CoverageGenerators(a) => tools::coverage_generators::run(a), + Check::GrammarCoverage(a) => tools::grammar_coverage::run(a), + Check::UiCoverage(a) => tools::ui_coverage::run(a), }; match result { Ok(code) => code, diff --git a/crates/mirth-lab/src/tools/coverage.rs b/crates/mirth-lab/src/tools/coverage.rs new file mode 100644 index 0000000..793bb0f --- /dev/null +++ b/crates/mirth-lab/src/tools/coverage.rs @@ -0,0 +1,111 @@ +//! Which of the compiler's functions ran, from a compiler built with rustc/coverage.toml. +//! +//! The site tables list each instrumented function (`cover` sites: id, crate, path, span); every +//! rustc process run with MIRTH_OUT set writes, at exit, a `V ` line for each function it +//! entered. This reads both and prints, per crate, how many functions ran; with --files, per +//! source file; with --unhit, the functions that never ran, in the crates or files matching. +//! Several --logs directories (several runs: fixtures, flags) are combined. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::PathBuf; +use std::process::ExitCode; + +use mirth_lab::coverage::{self, to_json_indent}; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + sites: PathBuf, + #[arg(long, required = true)] + logs: Vec, + #[arg(long)] + json: Option, + /// Also per source file. + #[arg(long)] + files: bool, + /// List the functions that never ran in crates or files matching. + #[arg(long)] + unhit: Vec, +} + +#[derive(Serialize, Default, Clone, Copy)] +struct Count { + functions: usize, + ran: usize, +} + +#[derive(Serialize)] +struct Report { + processes: usize, + functions: usize, + ran: usize, + crates: BTreeMap, + files: BTreeMap, + unhit: Vec, +} + +/// A span's file: all but its last two `:` fields. +pub fn span_file(span: &str) -> &str { + span.rsplitn(3, ':').last().unwrap_or(span) +} + +pub fn run(args: Args) -> anyhow::Result { + // A site listed twice counts once (the last listing). + let functions: Vec = + coverage::functions(&args.sites).into_iter().map(|f| (f.site.clone(), f)).collect::>().into_values().collect(); + let mut hit = BTreeSet::new(); + let mut processes = 0; + for d in &args.logs { + let (h, n) = coverage::hits(d); + hit.extend(h); + processes += n; + } + let sites: BTreeSet<&str> = functions.iter().map(|f| f.site.as_str()).collect(); + let unknown = hit.iter().filter(|s| !sites.contains(s.as_str())).count(); + let (mut by_crate, mut by_file) = (BTreeMap::::new(), BTreeMap::::new()); + for f in &functions { + let ran = hit.contains(&f.site) as usize; + for c in [by_crate.entry(f.krate.clone()).or_default(), by_file.entry(span_file(&f.span).to_owned()).or_default()] { + c.functions += 1; + c.ran += ran; + } + } + let total = sites.len(); + let ran = hit.iter().filter(|s| sites.contains(s.as_str())).count(); + print!("{processes} processes; {ran} of {total} functions ran ({:.1}%)", 100.0 * ran as f64 / total.max(1) as f64); + println!("{}", if unknown > 0 { format!("; {unknown} sites not in the tables") } else { String::new() }); + println!("{:40} {:>7} {:>7} {:>6}", "crate", "ran", "of", "%"); + let ratio = |c: &Count| c.ran as f64 / c.functions as f64; + let mut crates: Vec<(&String, &Count)> = by_crate.iter().collect(); + crates.sort_by(|a, b| ratio(a.1).total_cmp(&ratio(b.1))); + for (k, c) in crates { + println!("{k:40} {:7} {:7} {:6.1}", c.ran, c.functions, 100.0 * ratio(c)); + } + if args.files { + println!(); + println!("{:80} {:>6} {:>6}", "file", "ran", "of"); + let mut files: Vec<(&String, &Count)> = by_file.iter().collect(); + files.sort_by(|a, b| ratio(a.1).total_cmp(&ratio(b.1)).then(b.1.functions.cmp(&a.1.functions))); + for (k, c) in files { + println!("{k:80} {:6} {:6}", c.ran, c.functions); + } + } + let mut by_span: Vec<&coverage::Function> = functions.iter().collect(); + by_span.sort_by(|a, b| a.span.cmp(&b.span)); + for pattern in &args.unhit { + println!("\nnever ran, matching '{pattern}':"); + for f in &by_span { + if !hit.contains(&f.site) && (f.krate.contains(pattern.as_str()) || f.span.contains(pattern.as_str())) { + println!(" {} {}", f.path, f.span); + } + } + } + if let Some(j) = &args.json { + let mut unhit: Vec = functions.iter().filter(|f| !hit.contains(&f.site)).map(|f| format!("{}\t{}", f.path, f.span)).collect(); + unhit.sort(); + let report = Report { processes, functions: total, ran, crates: by_crate, files: by_file, unhit }; + std::fs::write(j, to_json_indent(&report, 1))?; + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/coverage_compact.rs b/crates/mirth-lab/src/tools/coverage_compact.rs new file mode 100644 index 0000000..8461eb2 --- /dev/null +++ b/crates/mirth-lab/src/tools/coverage_compact.rs @@ -0,0 +1,100 @@ +//! Fold coverage logs (MIRTH_OUT, from a compiler built with rustc/coverage.toml) into a running +//! union as they are finished, and delete them: a test suite run starts tens of thousands of rustc +//! processes, whose logs together would not fit on the disk. +//! +//! A log is finished when its last line is the `X` line written at exit, or when it has not +//! changed for ten minutes (a process that crashed). For each, /added.jsonl gets the +//! process's source file argument and the sites it reached that no earlier process did; +//! /union.txt holds every site reached so far, rewritten every pass. Runs until +//! exists, then does a last pass. + +use std::collections::BTreeSet; +use std::io::Write; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; +use std::time::{Duration, SystemTime}; + +use mirth_lab::coverage::{files_with, log_hits, to_json_line}; +use regex::Regex; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + logs: PathBuf, + #[arg(long)] + out: PathBuf, + #[arg(long, default_value = "")] + until: String, +} + +static CRATE_NAME: LazyLock = LazyLock::new(|| Regex::new(r"--crate-name\t(\S+)").unwrap()); + +/// The process's source file argument, from the log's header line, or its crate name. +fn source_of(header: &str) -> String { + if let Some(f) = header.split('\t').skip(3).find(|f| f.ends_with(".rs")) { + return f.to_owned(); + } + CRATE_NAME.captures(header).map(|c| c[1].to_owned()).unwrap_or_default() +} + +#[derive(serde::Serialize)] +struct Added<'a> { + source: String, + new: &'a [&'a str], +} + +fn one_pass(logs: &Path, out: &Path, union: &mut BTreeSet, last: bool) -> anyhow::Result { + let mut done = 0; + let mut added = std::fs::OpenOptions::new().create(true).append(true).open(out.join("added.jsonl"))?; + for log in files_with(logs, "log") { + let (Ok(bytes), Ok(meta)) = (std::fs::read(&log), std::fs::metadata(&log)) else { continue }; + let text = String::from_utf8_lossy(&bytes); + let age = meta.modified().ok().and_then(|m| SystemTime::now().duration_since(m).ok()).unwrap_or_default().as_secs_f64(); + let Some(last_line) = text.lines().last() else { continue }; + if !(last_line.starts_with("X\t") || age > 600.0 || (last && age > 5.0)) { + continue; + } + let sites: BTreeSet<&str> = log_hits(&text).collect(); + let new: Vec<&str> = sites.into_iter().filter(|s| !union.contains(*s)).collect(); + if !new.is_empty() { + let line = to_json_line(&Added { source: source_of(text.lines().next().unwrap_or("")), new: &new }); + writeln!(added, "{line}")?; + union.extend(new.iter().map(|s| s.to_string())); + } + let _ = std::fs::remove_file(&log); + done += 1; + } + let mut text = union.iter().map(String::as_str).collect::>().join("\n"); + text.push('\n'); + std::fs::write(out.join("union.txt"), text)?; + Ok(done) +} + +/// The local time as HH:MM:SS. +fn clock() -> String { + // SAFETY: time and localtime_r write only to the locals passed. + unsafe { + let t = libc::time(std::ptr::null_mut()); + let mut tm: libc::tm = std::mem::zeroed(); + libc::localtime_r(&t, &mut tm); + format!("{:02}:{:02}:{:02}", tm.tm_hour, tm.tm_min, tm.tm_sec) + } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.out)?; + let union_path = args.out.join("union.txt"); + let mut union: BTreeSet = std::fs::read_to_string(&union_path).unwrap_or_default().split_whitespace().map(str::to_owned).collect(); + loop { + let finishing = !args.until.is_empty() && Path::new(&args.until).exists(); + let n = one_pass(&args.logs, &args.out, &mut union, finishing)?; + println!("{} {n} logs folded, {} sites", clock(), union.len()); + if finishing { + one_pass(&args.logs, &args.out, &mut union, true)?; + break; + } + std::thread::sleep(Duration::from_secs(30)); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/coverage_generators.rs b/crates/mirth-lab/src/tools/coverage_generators.rs new file mode 100644 index 0000000..c9dd276 --- /dev/null +++ b/crates/mirth-lab/src/tools/coverage_generators.rs @@ -0,0 +1,309 @@ +//! Compiler runs that rustc's test suites hardly make, for coverage (`mirth-lab callgraph +//! --gaps` lists what is left): run them with the coverage-instrumented compiler and MIRTH_OUT +//! set, and fold the logs with `mirth-lab coverage-compact`. +//! +//! - prints: every `--print` request, on the host and on every target; +//! - targets: `tests/auxiliary/minicore.rs` and a file of functions with every kind of argument +//! and return value, compiled to an object for every target (each target's ABI, layout and +//! codegen code); +//! - links: for every target, a `no_main` binary, a cdylib, a staticlib and a dylib on minicore, +//! linked with `-Clinker=true` (the linker command each target's linker flavor builds, without +//! the linker), with linker options; +//! - dumps: each test of the list (`mirth-lab ui-coverage pick`) with each debugging and +//! printing option (`-Zunpretty=`, `-Zdump-mir`, `-Zprint-type-sizes`, statistics, profiling, ...). + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::time::Duration; + +use mirth_lab::rustc::run_command; +use mirth_lab::uitest; +use rayon::prelude::*; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// The rust checkout (tests/auxiliary/minicore.rs, tests/ui). + #[arg(long)] + rust: PathBuf, + /// The picked tests (picked.json of `mirth-lab ui-coverage pick`). + #[arg(long)] + list: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + jobs: usize, + #[arg(long, default_value = "prints,targets,dumps,links")] + only: String, +} + +const PRINTS: &[&str] = &[ + "all-target-specs-json", "backend-has-mnemonic", "backend-has-zstd", "calling-conventions", "cfg", "check-cfg", + "code-models", "crate-name", "crate-root-lint-levels", "deployment-target", "file-names", "host-tuple", "link-args", + "native-static-libs", "relocation-models", "split-debuginfo", "stack-protector-strategies", "supported-crate-types", + "sysroot", "target-cpus", "target-features", "target-libdir", "target-list", "target-spec-json", + "target-spec-json-schema", "tls-models", "wasm-proc-macro-tuple", +]; +const PER_TARGET: &[&str] = &[ + "cfg", "target-spec-json", "target-cpus", "target-features", "calling-conventions", "code-models", + "relocation-models", "tls-models", "stack-protector-strategies", "split-debuginfo", "supported-crate-types", + "deployment-target", "check-cfg", +]; + +/// Every kind of argument and return value, for each target's calling convention. +const ABI: &str = r#" +#![feature(no_core, lang_items, rustc_attrs, c_variadic, f16, f128)] +#![no_core] +#![crate_type = "lib"] +#![allow(improper_ctypes_definitions, unused)] +extern crate minicore; +use minicore::*; + +#[repr(C)] pub struct Small { a: u8, b: u16 } +#[repr(C)] pub struct Pair { a: u64, b: u64 } +#[repr(C)] pub struct Big { a: [u64; 8] } +#[repr(C)] pub struct Floats { a: f32, b: f64 } +#[repr(C)] pub struct Mixed { a: f32, b: u32 } +#[repr(C)] pub union U { a: u32, b: f32 } +#[repr(C)] pub struct Hfa { a: f32, b: f32, c: f32, d: f32 } +#[repr(C)] pub struct Empty {} +#[repr(transparent)] pub struct T(u64); +#[repr(C, packed)] pub struct Packed { a: u8, b: u32 } +#[repr(C, align(16))] pub struct Aligned { a: u8 } + +#[no_mangle] pub extern "C" fn c_small(x: Small) -> Small { x } +#[no_mangle] pub extern "C" fn c_pair(x: Pair) -> Pair { x } +#[no_mangle] pub extern "C" fn c_big(x: Big) -> Big { x } +#[no_mangle] pub extern "C" fn c_floats(x: Floats, y: f32, z: f64) -> Floats { x } +#[no_mangle] pub extern "C" fn c_mixed(x: Mixed) -> Mixed { x } +#[no_mangle] pub extern "C" fn c_union(x: U) -> U { x } +#[no_mangle] pub extern "C" fn c_hfa(x: Hfa) -> Hfa { x } +#[no_mangle] pub extern "C" fn c_empty(x: Empty) -> Empty { x } +#[no_mangle] pub extern "C" fn c_transparent(x: T) -> T { x } +#[no_mangle] pub extern "C" fn c_packed(x: Packed) -> Packed { x } +#[no_mangle] pub extern "C" fn c_aligned(x: Aligned) -> Aligned { x } +#[no_mangle] pub extern "C" fn c_ints(a: i8, b: u16, c: i32, d: u64, e: i128, f: u128, g: bool, h: char) -> i128 { e } +#[no_mangle] pub extern "C" fn c_ptrs(a: *const u8, b: &u32, c: &mut [u8; 3], f: extern "C" fn()) -> *const u8 { a } +#[no_mangle] pub extern "C" fn c_many(a: u64, b: u64, c: u64, d: u64, e: u64, f: u64, g: u64, h: u64, i: u64, j: Pair, k: f64, l: f64, m: f64, n: f64, o: f64, p: f64, q: f64, r: f64, s: f64) -> u64 { a } +#[no_mangle] pub unsafe extern "C" fn c_variadic(a: u32, mut args: ...) -> u32 { a } +pub fn rust_all(a: Small, b: Pair, c: Big, d: Floats, e: (u8, u64), f: [u32; 5], g: &[u8], h: &str, i: u128) -> Big { c } +pub fn rust_f16(a: f16, b: f128) -> f128 { b } +#[no_mangle] pub extern "C" fn c_f16(a: f16, b: f128) -> f128 { b } +#[no_mangle] pub extern "system" fn system(a: Pair) -> Pair { a } +#[no_mangle] pub extern "C-unwind" fn c_unwind(a: Pair) -> Pair { a } +pub static TABLE: [extern "C" fn(Pair) -> Pair; 2] = [c_pair, c_unwind_shim]; +extern "C" fn c_unwind_shim(a: Pair) -> Pair { a } +extern "C" { fn imported(a: Big, b: Floats) -> Hfa; } +pub unsafe fn call_imported(a: Big, b: Floats) -> Hfa { imported(a, b) } +"#; + +const LINKED: &str = r#" +#![feature(no_core, lang_items)] +#![no_core] +#![no_main] +extern crate minicore; +#[no_mangle] pub extern "C" fn exported(a: u32) -> u32 { a } +#[no_mangle] pub static DATA: u32 = 7; +#[link(name = "c")] extern "C" { fn puts(p: *const u8) -> i32; } +#[link(name = "m", kind = "static")] extern "C" {} +#[link(name = "framework_like", kind = "dylib", modifiers = "+verbatim")] extern "C" {} +"#; + +const LINK_OPTIONS: &[&[&str]] = &[ + &[], &["-Cprefer-dynamic", "-Crelocation-model=pic"], &["-Cstrip=symbols", "-Clink-dead-code"], + &["-Clink-self-contained=yes"], &["-Cdebuginfo=2", "-Csplit-debuginfo=packed"], + &["-Clink-arg=-Wl,--foo", "-Clink-args=-x -y", "-Zpre-link-args=-z"], + &["-Cdefault-linker-libraries", "-Zlink-native-libraries=no"], &["-Ccontrol-flow-guard"], + &["-Zstaticlib-allow-rdylib-deps"], &["-Copt-level=s", "-Clto=fat"], &["-Ccode-model=large"], +]; + +const UNPRETTY: &[&str] = &[ + "normal", "expanded", "expanded,identified", "expanded,hygiene", "ast-tree", "ast-tree,expanded", "hir", + "hir,identified", "hir,typed", "hir-tree", "thir-tree", "thir-flat", "mir", "stable-mir", "mir-cfg", +]; +const DUMPS: &[&[&str]] = &[ + &["-Zdump-mir=all", "-Zdump-mir-dataflow", "-Zdump-mir-graphviz", "-Zmir-include-spans=on"], + &["-Zprint-type-sizes"], &["-Zprint-mono-items=yes", "--emit=link"], &["-Zmeta-stats"], &["-Zhir-stats"], + &["-Zinput-stats"], &["-Zself-profile", "-Zself-profile-events=all"], &["-Ztime-passes"], + &["-Zquery-dep-graph", "-Zdump-dep-graph", "-Cincremental=inc"], &["-Zincremental-info", "-Cincremental=inc"], + &["-Zdump-mono-stats", "-Zdump-mono-stats-format=json", "--emit=link"], &["-Zprint-codegen-stats", "--emit=link"], + &["-Zvalidate-mir", "-Zlint-mir", "-Zmir-opt-level=4"], &["-Zverbose-internals", "-Zidentify-regions"], + &["-Ztrack-diagnostics", "-Zteach"], &["-Zthreads=4"], &["-Zpolonius=next"], &["-Zinline-mir", "-Zmir-opt-level=3"], + &["-Zrandomize-layout"], &["-Zwrite-long-types-to-disk=no", "-Zverbose-internals"], + &["-Zunleash-the-miri-inside-of-you"], &["-Zno-analysis"], &["-Zprofile-closures"], &["-Zui-testing"], + &["-Cinstrument-coverage", "--emit=link"], &["-Zemit-stack-sizes", "--emit=link"], + &["--error-format=json", "--json=diagnostic-rendered-ansi,artifacts,future-incompat,unused-externs"], + &["--error-format=human-annotate-rs"], &["--error-format=short"], &["-Zterminal-urls=yes", "--color=always"], + &["-Wunused", "-Wrust-2018-idioms", "-Wrust-2021-compatibility", "-Wrust-2024-compatibility", "-Wclippy::all"], + &["-Fwarnings", "--cap-lints=warn"], &["-Zcodegen-source-order", "--emit=link"], +]; + +/// The debugging and printing options: each `-Zunpretty=` mode, then the rest. +fn dumps_options() -> Vec> { + let unpretty = UNPRETTY.iter().map(|m| vec![format!("-Zunpretty={m}")]); + unpretty.chain(DUMPS.iter().map(|o| o.iter().map(|s| s.to_string()).collect())).collect() +} + +/// Run with RUSTC_BOOTSTRAP; true when it exits 0. +fn ok>(program: &Path, argv: &[S], cwd: &Path, secs: u64) -> bool { + let mut cmd = Command::new(program); + cmd.args(argv).current_dir(cwd).env("RUSTC_BOOTSTRAP", "1"); + run_command(cmd, Duration::from_secs(secs)).is_ok_and(|f| f.success()) +} + +fn targets(args: &Args) -> Vec { + let out = Command::new(&args.rustc).args(["--print", "target-list"]).env("RUSTC_BOOTSTRAP", "1").output(); + out.map(|o| String::from_utf8_lossy(&o.stdout).split_whitespace().map(str::to_owned).collect()).unwrap_or_default() +} + +fn scratch(args: &Args) -> tempfile::TempDir { + tempfile::tempdir_in(&args.work).expect("scratch directory") +} + +fn prints(args: &Args) { + let d = scratch(args); + let empty = d.path().join("lib.rs"); + let _ = std::fs::write(&empty, ""); + let empty = empty.to_string_lossy().into_owned(); + let mut jobs: Vec> = PRINTS.iter().map(|k| vec!["--print".into(), k.to_string(), "-Zunstable-options".into(), empty.clone()]).collect(); + for target in targets(args) { + for k in PER_TARGET { + jobs.push(vec!["--print".into(), k.to_string(), "--target".into(), target.clone(), "-Zunstable-options".into(), empty.clone()]); + } + } + let succeeded = jobs.par_iter().filter(|j| ok(&args.rustc, j, d.path(), 300)).count(); + println!("prints: {} runs, {succeeded} succeeded", jobs.len()); +} + +/// The common arguments for a `no_core` build for `target` into `d`. +fn target_base(target: &str, d: &Path) -> Vec { + ["--target", target, "-Zunstable-options", "--edition", "2021", "-Cpanic=abort", "--out-dir"] + .iter() + .map(|s| s.to_string()) + .chain([d.to_string_lossy().into_owned()]) + .collect() +} + +fn one_target(args: &Args, target: &str) -> Vec { + let d = scratch(args); + let p = d.path(); + let _ = std::fs::write(p.join("abi.rs"), ABI); + let minicore = args.rust.join("tests/auxiliary/minicore.rs").to_string_lossy().into_owned(); + let base = target_base(target, p); + let code = |extra: &[String]| { + let mut cmd = Command::new(&args.rustc); + cmd.args(&base).args(extra).current_dir(p).env("RUSTC_BOOTSTRAP", "1"); + match run_command(cmd, Duration::from_secs(300)) { + Ok(f) => match f.exit { + mirth_lab::rustc::Exit::Code(c) => c, + _ => -1, + }, + Err(_) => -1, + } + }; + let s = |v: &[&str]| v.iter().map(|x| x.to_string()).collect::>(); + let mut results = vec![code(&[s(&["--crate-type", "rlib", "--crate-name", "minicore", "-Copt-level=1", "--emit=link,obj"]), vec![minicore]].concat())]; + if results[0] == 0 { + let rlib = format!("minicore={}/libminicore.rlib", p.display()); + for opt in ["0", "3"] { + results.push(code(&[s(&["--emit=obj,asm,llvm-ir"]), vec![format!("-Copt-level={opt}")], s(&["-Cdebuginfo=2", "--extern"]), vec![rlib.clone()], s(&["abi.rs"])].concat())); + } + } + results +} + +fn cross(args: &Args) -> anyhow::Result<()> { + let done: Vec<(String, Vec)> = targets(args).into_par_iter().map(|t| { let r = one_target(args, &t); (t, r) }).collect(); + let built = done.iter().filter(|(_, r)| r.len() == 3 && r.iter().all(|c| *c == 0)).count(); + println!("targets: {} targets, {built} built minicore and the ABI file", done.len()); + let map: BTreeMap> = done.into_iter().collect(); + std::fs::write(args.work.join("targets.json"), mirth_lab::coverage::to_json_indent(&map, 0))?; + Ok(()) +} + +fn one_link(args: &Args, target: &str) -> usize { + let d = scratch(args); + let p = d.path(); + let _ = std::fs::write(p.join("linked.rs"), LINKED); + let minicore = args.rust.join("tests/auxiliary/minicore.rs").to_string_lossy().into_owned(); + let mut base = target_base(target, p); + base.push("-Clinker=true".into()); + let with = |extra: &[&str]| -> Vec { base.iter().cloned().chain(extra.iter().map(|s| s.to_string())).collect() }; + let mut minicore_argv = with(&["--crate-type", "rlib", "--crate-name", "minicore", "--emit=link"]); + minicore_argv.push(minicore); + if !ok(&args.rustc, &minicore_argv, p, 300) { + return 0; + } + let rlib = format!("minicore={}/libminicore.rlib", p.display()); + let mut n = 0; + for options in LINK_OPTIONS { + for kind in ["bin", "cdylib", "staticlib", "dylib"] { + let mut argv = with(&["--crate-type", kind, "--extern", &rlib, "-Csave-temps"]); + argv.extend(options.iter().map(|s| s.to_string())); + argv.push("linked.rs".into()); + n += ok(&args.rustc, &argv, p, 300) as usize; + } + } + n +} + +fn links(args: &Args) { + let done: Vec = targets(args).par_iter().map(|t| one_link(args, t)).collect(); + println!("links: {} targets, {} links succeeded", done.len(), done.iter().sum::()); +} + +fn dump(args: &Args, options: &[Vec], test: &str) -> usize { + let path = args.rust.join("tests/ui").join(test); + let text = String::from_utf8_lossy(&std::fs::read(&path).unwrap_or_default()).into_owned(); + let (flags, edition, _, _) = uitest::headers(&text); + let mut n = 0; + for extra in options { + let d = scratch(args); + let mut argv: Vec = vec![path.to_string_lossy().into_owned(), "--edition".into(), edition.clone().unwrap_or_else(|| "2015".into())]; + if !extra.iter().any(|e| e.starts_with("--emit")) { + argv.push("--emit=metadata".into()); + } + argv.extend(["--out-dir".into(), d.path().to_string_lossy().into_owned()]); + argv.extend(["-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"].map(String::from)); + argv.extend(flags.iter().cloned()); + argv.extend(extra.iter().cloned()); + n += ok(&args.rustc, &argv, d.path(), 120) as usize; + } + n +} + +fn dumps(args: &Args) -> anyhow::Result<()> { + let picked: Vec = serde_json::from_str(&std::fs::read_to_string(&args.list)?)?; + let tests: Vec = picked + .iter() + .filter_map(|t| t.get("test").unwrap_or(t).as_str().map(str::to_owned)) + .collect(); + let options = dumps_options(); + let succeeded: usize = tests.par_iter().map(|t| dump(args, &options, t)).sum(); + println!("dumps: {} tests x {} options, {succeeded} runs succeeded", tests.len(), options.len()); + Ok(()) +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let only: Vec<&str> = args.only.split(',').collect(); + pool.install(|| -> anyhow::Result<()> { + if only.contains(&"prints") { + prints(&args); + } + if only.contains(&"targets") { + cross(&args)?; + } + if only.contains(&"dumps") { + dumps(&args)?; + } + if only.contains(&"links") { + links(&args); + } + Ok(()) + })?; + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/grammar_coverage.rs b/crates/mirth-lab/src/tools/grammar_coverage.rs new file mode 100644 index 0000000..2a079ee --- /dev/null +++ b/crates/mirth-lab/src/tools/grammar_coverage.rs @@ -0,0 +1,268 @@ +//! Which parts of the Rust grammar a fixture's sources use, by Ur's grammar. +//! +//! Ur's Rust grammar (Urscal modules, `syntax Sort = Label: ... | Label: ... | Other ;`) names +//! each alternative of each syntax sort. `ur parse --tree` prints a file's tree with every node as +//! `(Sort::Label ...)` and every literal token in quotes. Two measures: +//! +//! alternatives each labeled alternative, by construct: the label within its module, with +//! Conditions.rsc (expressions in condition position) counted as Expressions +//! literals each keyword and operator the syntax rules mention, used or not +//! +//! Prints what is missing, by module, and a summary. Lexical rules, layout and keyword lists are +//! not syntax alternatives and are left out. + +use std::collections::{BTreeMap, BTreeSet, HashMap}; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; + +use mirth_lab::coverage::to_json_indent; +use regex::Regex; +use serde::Serialize; +use walkdir::WalkDir; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + ur: PathBuf, + /// Ur's Rust grammar: /ecosystems/rust/language. + #[arg(long)] + grammar: PathBuf, + fixture: PathBuf, + #[arg(long)] + json: Option, +} + +const STRING: &str = r#""(?:[^"\\]|\\.)*""#; +static STRING_RE: LazyLock = LazyLock::new(|| Regex::new(STRING).unwrap()); +static STRING_AT: LazyLock = LazyLock::new(|| Regex::new(&format!("^{STRING}")).unwrap()); +static COMMENT: LazyLock = LazyLock::new(|| Regex::new(r"//[^\n]*").unwrap()); +static SYNTAX: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^syntax\s+(\w+)[^=]*=").unwrap()); +static ATTR_STRING: LazyLock = LazyLock::new(|| Regex::new(&format!(r"@\w+={STRING}")).unwrap()); +static ATTR_EMPTY: LazyLock = LazyLock::new(|| Regex::new(r#"@\w+="[^"]*""#).unwrap()); +static NODE: LazyLock = LazyLock::new(|| Regex::new(r"\((\w+)::(\w+)").unwrap()); + +/// Conditions.rsc repeats the expression sorts for condition position (no struct literals): the +/// same constructs, so counted with the expressions. +fn family(stem: &str) -> &str { + if stem == "Conditions" { "Expressions" } else { stem } +} + +fn is_word(c: char) -> bool { + c.is_alphanumeric() || c == '_' +} + +/// Whether the character before byte `i` is neither a word character nor `:`. +fn free_before(text: &str, i: usize) -> bool { + text[..i].chars().next_back().is_none_or(|c| !is_word(c) && c != ':') +} + +/// (sort, body) for each `syntax` rule of a module: from `syntax Name ... =` to the `;` that +/// ends it, outside quotes and brackets. +fn syntax_rules(text: &str) -> Vec<(String, String)> { + let bytes = text.as_bytes(); + SYNTAX + .captures_iter(text) + .map(|m| { + let start = m.get(0).unwrap().end(); + let (mut i, mut depth) = (start, 0i32); + while i < bytes.len() { + match bytes[i] { + b'"' => { + i += STRING_AT.find(&text[i..]).map_or(1, |q| q.end()); + continue; + } + b'(' | b'[' | b'{' => depth += 1, + b')' | b']' | b'}' => depth -= 1, + b';' if depth == 0 => break, + _ => {} + } + i += 1; + } + (m[1].to_owned(), text[start..i.min(bytes.len())].to_owned()) + }) + .collect() +} + +/// The labels of a rule body: `Label:` (not `Label::`) not preceded by a word character or `:`. +fn labels(body: &str) -> Vec<&str> { + let mut out = Vec::new(); + let mut i = 0; + while i < body.len() { + let c = body[i..].chars().next().unwrap(); + if c.is_ascii_uppercase() && free_before(body, i) { + let end = body[i..].find(|c: char| !is_word(c)).map_or(body.len(), |e| i + e); + let colon = body[end..].find(|c: char| !c.is_whitespace()).map_or(body.len(), |e| end + e); + if body[colon..].starts_with(':') && !body[colon + 1..].starts_with(':') { + out.push(&body[i..end]); + i = colon + 1; + continue; + } + } + i += c.len_utf8(); + } + out +} + +/// The quoted tokens of a tree: a string not preceded by a word character or `:`. +fn quoted(tree: &str) -> Vec<&str> { + let mut out = Vec::new(); + let mut i = 0; + while let Some(q) = tree[i..].find('"').map(|q| i + q) { + if free_before(tree, q) + && let Some(m) = STRING_AT.find(&tree[q..]) + { + out.push(&tree[q + 1..q + m.end() - 1]); + i = q + m.end(); + } else { + i = q + 1; + } + } + out +} + +/// Python's repr of a string. +fn py_repr(s: &str) -> String { + let quote = if s.contains('\'') && !s.contains('"') { '"' } else { '\'' }; + let mut out = String::from(quote); + for c in s.chars() { + match c { + '\\' => out.push_str("\\\\"), + '\n' => out.push_str("\\n"), + '\t' => out.push_str("\\t"), + '\r' => out.push_str("\\r"), + c if c == quote => { + out.push('\\'); + out.push(c); + } + c if (c as u32) < 0x20 || c as u32 == 0x7f => out.push_str(&format!("\\x{:02x}", c as u32)), + c => out.push(c), + } + } + out.push(quote); + out +} + +fn modules(grammar: &Path) -> Vec { + let mut files: Vec = mirth_lab::coverage::files_with(grammar, "rsc"); + files.sort(); + files +} + +#[derive(Serialize)] +struct Report { + alternatives: BTreeMap, + literals: BTreeMap, + failed: Vec, +} + +pub fn run(args: Args) -> anyhow::Result { + let mut grammar: BTreeMap<(String, String), Vec> = BTreeMap::new(); + let mut literals: HashMap = HashMap::new(); + let mut sort_family: HashMap = HashMap::new(); + for f in modules(&args.grammar) { + let stem = f.file_stem().and_then(|s| s.to_str()).unwrap_or("").to_owned(); + let fam = family(&stem).to_owned(); + let text = COMMENT.replace_all(&std::fs::read_to_string(&f)?, "").into_owned(); + let rules = syntax_rules(&text); + for (sort, _) in &rules { + sort_family.entry(sort.clone()).or_insert_with(|| fam.clone()); + } + if ["Testing", "Semantics", "Language", "Rust"].contains(&stem.as_str()) { + continue; + } + for (sort, body) in rules { + let body = ATTR_STRING.replace_all(&body, ""); + for lit in STRING_RE.find_iter(&body) { + let lit = &lit.as_str()[1..lit.as_str().len() - 1]; + let lit = lit.replace("\\<", "<").replace("\\>", ">").replace("\\\"", "\"").replace("\\\\", "\\"); + literals.entry(lit).or_insert_with(|| fam.clone()); + } + let bare = STRING_RE.replace_all(&body, "\"\""); + let bare = ATTR_EMPTY.replace_all(&bare, ""); + for label in labels(&bare) { + grammar.entry((fam.clone(), label.to_owned())).or_default().push(sort.clone()); + } + } + } + let mut files: Vec = WalkDir::new(&args.fixture) + .into_iter() + .filter_map(Result::ok) + .map(|e| e.into_path()) + .filter(|p| p.extension().is_some_and(|x| x == "rs") && !p.components().any(|c| c.as_os_str() == "target")) + .collect(); + files.sort(); + let mut used_alts: HashMap<(String, String), usize> = HashMap::new(); + let mut used_lits: HashMap = HashMap::new(); + let mut failed = Vec::new(); + for f in &files { + let out = Command::new(&args.ur).args(["parse", "--tree"]).arg(f).output()?; + let tree = String::from_utf8_lossy(&out.stdout); + if !out.status.success() || !tree.contains("(File::") { + failed.push(f.display().to_string()); + continue; + } + for c in NODE.captures_iter(&tree) { + let fam = sort_family.get(&c[1]).cloned().unwrap_or_else(|| c[1].to_owned()); + *used_alts.entry((fam, c[2].to_owned())).or_default() += 1; + } + for lit in quoted(&tree) { + *used_lits.entry(lit.replace("\\\"", "\"").replace("\\\\", "\\")).or_default() += 1; + } + } + let mut by_module: BTreeMap<&str, Vec> = BTreeMap::new(); + let mut missing_alts = 0; + for ((fam, label), sorts) in &grammar { + if !used_alts.contains_key(&(fam.clone(), label.clone())) { + let sorts: BTreeSet<&String> = sorts.iter().collect(); + by_module.entry(fam).or_default().push(format!("{label} ({})", sorts.into_iter().cloned().collect::>().join("/"))); + missing_alts += 1; + } + } + let mut missing_lits: Vec<(&String, &String)> = literals.iter().filter(|(l, _)| !used_lits.contains_key(*l)).map(|(l, m)| (m, l)).collect(); + missing_lits.sort(); + let mut lit_by_module: BTreeMap<&str, Vec> = BTreeMap::new(); + for (m, lit) in &missing_lits { + lit_by_module.entry(m).or_default().push(py_repr(lit)); + } + let names: BTreeSet<&str> = by_module.keys().chain(lit_by_module.keys()).copied().collect(); + for m in names { + println!("== {m}"); + if let Some(a) = by_module.get(m) { + println!(" alternatives: {}", a.join(", ")); + } + if let Some(l) = lit_by_module.get(m) { + println!(" literals: {}", l.join(" ")); + } + } + let failed_list = if failed.is_empty() { + String::new() + } else { + format!(": [{}]", failed.iter().map(|f| py_repr(f)).collect::>().join(", ")) + }; + println!("{} files, {} failed to parse{failed_list}", files.len(), failed.len()); + println!("alternatives: {} of {} used", grammar.len() - missing_alts, grammar.len()); + println!("literals: {} of {} used", literals.len() - missing_lits.len(), literals.len()); + if let Some(j) = &args.json { + let report = Report { + alternatives: grammar.keys().map(|(f, l)| (format!("{f}::{l}"), used_alts.get(&(f.clone(), l.clone())).copied().unwrap_or(0))).collect(), + literals: literals.keys().map(|l| (l.clone(), used_lits.get(l).copied().unwrap_or(0))).collect(), + failed, + }; + std::fs::write(j, to_json_indent(&report, 1))?; + } + Ok(ExitCode::SUCCESS) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn labels_and_quotes() { + assert_eq!(labels("A: x | Foo::Bar | bC: y | Baz : z | (Q:: r) | Last:"), ["A", "Baz", "Last"]); + assert_eq!(quoted(r#"(X "a" b"c" "d\"e" :"f")"#), ["a", r#"d\"e"#]); + assert_eq!(py_repr("it's"), "\"it's\""); + assert_eq!(py_repr("a\\b"), "'a\\\\b'"); + } +} diff --git a/crates/mirth-lab/src/tools/ui_coverage.rs b/crates/mirth-lab/src/tools/ui_coverage.rs new file mode 100644 index 0000000..5d9a61f --- /dev/null +++ b/crates/mirth-lab/src/tools/ui_coverage.rs @@ -0,0 +1,270 @@ +//! Which compiler functions each of rustc's UI tests reaches that a baseline (the fixture's +//! builds) does not, with a coverage-instrumented rustc (rustc/coverage.toml); then a small set +//! of tests that reaches the most of them. +//! +//! `run` compiles each test file the way its `//@` headers say, as far as one rustc call can: +//! `compile-flags`, `edition`, the first of `revisions` (as `--cfg` with its own flags), metadata +//! only for tests that do not build (check-pass, and tests expected to fail before codegen), a +//! full build otherwise. Tests that need auxiliary crates, proc macros, another target or +//! `minicore` are skipped. Writes /tests.jsonl: per test, whether it compiled and the +//! indices (into /functions.json) of the functions it reached beyond the baseline. +//! +//! `pick` chooses tests greedily, each adding the most functions not yet reached, and writes +//! /picked.json. + +use std::collections::{BTreeMap, BTreeSet, HashSet}; +use std::io::Write; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::{LazyLock, Mutex}; +use std::time::Duration; + +use mirth_lab::coverage::{self, to_json_indent, to_json_line}; +use mirth_lab::rustc::{is_ice, run_command, Exit}; +use mirth_lab::uitest::{self, Kind}; +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; +use walkdir::WalkDir; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[command(subcommand)] + cmd: Cmd, +} + +#[derive(clap::Subcommand, Debug)] +enum Cmd { + /// Compile each UI test; record the functions it reaches beyond the baseline. + Run(RunArgs), + /// Choose the tests that together reach the most. + Pick(PickArgs), +} + +#[derive(clap::Args, Debug)] +struct RunArgs { + #[arg(long)] + rustc: PathBuf, + #[arg(long)] + tests: PathBuf, + #[arg(long)] + sites: PathBuf, + /// Log directories (or directories of them) of the baseline. + #[arg(long)] + baseline: Vec, + #[arg(long)] + out: PathBuf, + #[arg(long, default_value_t = 6)] + jobs: usize, + #[arg(long, default_value_t = 0)] + limit: usize, +} + +#[derive(clap::Args, Debug)] +struct PickArgs { + #[arg(long)] + out: PathBuf, + #[arg(long, default_value_t = 200)] + count: usize, +} + +static SKIP: LazyLock = LazyLock::new(|| { + Regex::new(r"(?m)^//@\s*(aux-build|aux-crate|aux-bin|proc-macro|add-minicore|needs-llvm-components|only-|ignore-x86_64|ignore-linux|needs-sanitizer|needs-profiler|needs-asm-support|known-bug)").unwrap() +}); +/// `only-` directives this host satisfies. +const HOST_ONLY: &[&str] = &["x86_64", "linux", "unix", "64bit"]; + +fn skipped(text: &str) -> bool { + SKIP.captures_iter(text).any(|c| { + let m = c.get(1).unwrap(); + m.as_str() != "only-" || !HOST_ONLY.iter().any(|h| text[m.end()..].starts_with(h)) + }) +} + +#[derive(Serialize)] +#[serde(untagged)] +enum Outcome { + Ran { test: String, status: String, kind: Option, error: String, new: Vec }, + Skipped { test: String, status: String }, +} + +#[derive(Serialize)] +struct Functions<'a> { + functions: Vec<[&'a str; 2]>, + baseline: Vec, +} + +fn run_one(a: &RunArgs, path: &Path, index: &BTreeMap, baseline: &HashSet) -> Outcome { + let text = String::from_utf8_lossy(&std::fs::read(path).unwrap_or_default()).into_owned(); + let rel = path.strip_prefix(&a.tests).unwrap_or(path).to_string_lossy().into_owned(); + if skipped(&text) { + return Outcome::Skipped { test: rel, status: "skipped".into() }; + } + let (flags, edition, kind, _) = uitest::headers(&text); + let d = tempfile::tempdir_in(a.out.join("scratch")).expect("scratch directory"); + let emit = if Kind::is_check(kind) { "--emit=metadata" } else { "--emit=link" }; + let logs = d.path().join("logs"); + let mut cmd = Command::new(&a.rustc); + cmd.arg(path) + .args(["--edition", edition.as_deref().unwrap_or("2015"), emit, "--out-dir"]) + .arg(d.path()) + .args(["-Zunstable-options", "-Ainternal_features"]) + .args(&flags) + .env("MIRTH_OUT", &logs) + .env("RUSTC_BOOTSTRAP", "1") + .current_dir(d.path()); + let (status, error) = match run_command(cmd, Duration::from_secs(120)) { + Ok(f) if f.exit == Exit::Timeout => ("timeout".to_owned(), String::new()), + Ok(f) => { + let stderr = f.stderr_text(); + let status = if f.success() { "ok" } else if is_ice(&stderr) { "ice" } else { "error" }; + let first: String = stderr.lines().find(|l| l.starts_with("error")).unwrap_or("").chars().take(160).collect(); + (status.to_owned(), first) + } + Err(e) => ("error".to_owned(), e.to_string()), + }; + let (hit, _) = coverage::hits(&logs); + let mut new: Vec = hit.iter().filter(|s| !baseline.contains(*s)).filter_map(|s| index.get(s).copied()).collect(); + new.sort(); + Outcome::Ran { test: rel, status, kind, error, new } +} + +fn run_tests(a: RunArgs) -> anyhow::Result<()> { + let mut functions: Vec<(String, String, String)> = coverage::functions(&a.sites).into_iter().map(|f| (f.site, f.path, f.span)).collect(); + functions.sort(); + let index: BTreeMap = functions.iter().enumerate().map(|(i, f)| (f.0.clone(), i)).collect(); + let mut baseline: HashSet = HashSet::new(); + for b in &a.baseline { + let mut dirs = vec![b.clone()]; + if let Ok(entries) = std::fs::read_dir(b) { + dirs.extend(entries.flatten().map(|e| e.path()).filter(|p| !p.file_name().is_some_and(|n| n.to_string_lossy().starts_with('.')))); + } + for d in dirs.iter().filter(|d| d.is_dir()) { + baseline.extend(coverage::hits(d).0); + } + } + std::fs::create_dir_all(a.out.join("scratch"))?; + let mut base_idx: Vec = baseline.iter().filter_map(|s| index.get(s).copied()).collect(); + base_idx.sort(); + let info = Functions { functions: functions.iter().map(|f| [f.1.as_str(), f.2.as_str()]).collect(), baseline: base_idx.clone() }; + std::fs::write(a.out.join("functions.json"), to_json_line(&info))?; + let results = a.out.join("tests.jsonl"); + let done: HashSet = std::fs::read_to_string(&results) + .unwrap_or_default() + .lines() + .filter_map(|l| serde_json::from_str::(l).ok()) + .filter_map(|v| v["test"].as_str().map(str::to_owned)) + .collect(); + let mut tests: Vec = WalkDir::new(&a.tests) + .into_iter() + .filter_map(Result::ok) + .map(|e| e.into_path()) + .filter(|p| p.extension().is_some_and(|x| x == "rs") && p.is_file()) + .filter(|p| { + let rel = p.strip_prefix(&a.tests).unwrap_or(p); + !rel.components().any(|c| c.as_os_str() == "auxiliary") && !done.contains(&*rel.to_string_lossy()) + }) + .collect(); + tests.sort(); + if a.limit > 0 { + tests.truncate(a.limit); + } + println!("{} functions, {} in the baseline; {} tests to run", functions.len(), base_idx.len(), tests.len()); + let out = Mutex::new(std::fs::OpenOptions::new().create(true).append(true).open(&results)?); + let n = AtomicUsize::new(0); + let pool = rayon::ThreadPoolBuilder::new().num_threads(a.jobs).build()?; + pool.install(|| { + tests.par_iter().for_each(|t| { + let res = run_one(&a, t, &index, &baseline); + let mut f = out.lock().unwrap(); + let _ = writeln!(f, "{}", to_json_line(&res)); + let k = n.fetch_add(1, Ordering::Relaxed); + if k % 500 == 0 { + let _ = f.flush(); + println!("{k} tests"); + } + }) + }); + let _ = std::fs::remove_dir_all(a.out.join("scratch")); + Ok(()) +} + +/// What `pick` reads of a line of tests.jsonl. +#[derive(Deserialize)] +struct Line { + test: String, + status: String, + #[serde(default)] + new: Vec, +} + +#[derive(Deserialize)] +struct Info { + functions: Vec, + baseline: Vec, +} + +#[derive(Serialize)] +struct Picked { + test: String, + status: String, + adds: usize, + total: usize, +} + +fn pick(a: PickArgs) -> anyhow::Result<()> { + let info: Info = serde_json::from_str(&std::fs::read_to_string(a.out.join("functions.json"))?)?; + // In file order: a tie goes to the test listed first, as Python's max does. + let mut sets: Vec<(String, BTreeSet)> = Vec::new(); + let mut status: BTreeMap = BTreeMap::new(); + let mut reached = BTreeSet::new(); + for line in std::fs::read_to_string(a.out.join("tests.jsonl"))?.lines() { + let Ok(Line { test, status: s, new }) = serde_json::from_str(line) else { continue }; + if new.is_empty() { + continue; + } + reached.extend(new.iter().copied()); + let set: BTreeSet = new.into_iter().collect(); + match sets.iter_mut().find(|(t, _)| *t == test) { + Some(entry) => entry.1 = set, + None => sets.push((test.clone(), set)), + } + status.insert(test, s); + } + let mut covered: BTreeSet = BTreeSet::new(); + let mut picked = Vec::new(); + while picked.len() < a.count && !sets.is_empty() { + let gains: Vec = sets.iter().map(|(_, s)| s.difference(&covered).count()).collect(); + let best = (0..sets.len()).fold(0, |b, i| if gains[i] > gains[b] { i } else { b }); + if gains[best] == 0 { + break; + } + let (test, set) = sets.remove(best); + covered.extend(set); + picked.push(Picked { status: status[&test].clone(), test, adds: gains[best], total: covered.len() }); + } + let total = info.functions.len(); + let base = info.baseline.len(); + println!( + "baseline {base} of {total} functions ({:.1}%); all tests reach {} more; {} picked tests reach {} more ({:.1}% in all)", + 100.0 * base as f64 / total as f64, + reached.len(), + picked.len(), + covered.len(), + 100.0 * (base + covered.len()) as f64 / total as f64 + ); + for t in picked.iter().take(40) { + println!(" +{:5} {:6} {:7} {}", t.adds, t.total, t.status, t.test); + } + std::fs::write(a.out.join("picked.json"), to_json_indent(&picked, 1))?; + Ok(()) +} + +pub fn run(args: Args) -> anyhow::Result { + match args.cmd { + Cmd::Run(a) => run_tests(a)?, + Cmd::Pick(a) => pick(a)?, + } + Ok(ExitCode::SUCCESS) +} diff --git a/rustc/coverage-fulldeps.sh b/rustc/coverage-fulldeps.sh index 53fd8f1..9bfdb0e 100755 --- a/rustc/coverage-fulldeps.sh +++ b/rustc/coverage-fulldeps.sh @@ -11,7 +11,9 @@ host=x86_64-unknown-linux-gnu out=$1/ui-fulldeps-stage1 mkdir -p "$out/logs" "$out/bin" rm -f "$out/done" -python3 "$here/coverage-compact.py" --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & +# The compactor is a mirth-lab subcommand: build it first (a no-op when up to date). +(cd "$here/.." && cargo build --release -q --offline -p mirth-lab) || exit 1 +"$here/../target/release/mirth-lab" coverage-compact --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & compactor=$! pass=0; fail=0 for t in $(grep -l '^//@ ignore-stage1' -r "$MIRTH_RUST/tests/ui-fulldeps" --include=*.rs | grep -v /auxiliary/ | sort); do diff --git a/rustc/coverage-report.sh b/rustc/coverage-report.sh index 0a8a644..8a4ead9 100755 --- a/rustc/coverage-report.sh +++ b/rustc/coverage-report.sh @@ -34,6 +34,7 @@ dirs=${COV_LOGS-$work/cov-sink $work/cov-flags} for d in $dirs; do [ -d "$d" ] && logs+=(--logs "$d") done +(cd "$here/.." && cargo build --release -q --offline -p mirth-lab) || exit 1 "$here/../target/release/mirth-lab" callgraph --graph "$work/build-cg/mirth-sites" --sites "$build/mirth-sites" \ "${runs[@]}" "${logs[@]}" --gaps "$report/gaps.md" --block-gaps "$report/gaps-blocks.md" --json "$report/gaps.json" "$@" \ > "$report/coverage-report.txt" diff --git a/rustc/coverage-run.sh b/rustc/coverage-run.sh index 77a09a5..6be75e8 100755 --- a/rustc/coverage-run.sh +++ b/rustc/coverage-run.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # Run any command with MIRTH_OUT set, folding the coverage logs of the instrumented compiler's -# processes into $WORK/cov-suites//union.txt as they finish (rustc/coverage-compact.py), +# processes into $WORK/cov-suites//union.txt as they finish (`mirth-lab coverage-compact`), # where rustc/coverage-report.sh picks them up (COV_SUITES: another directory instead). # # WORK=~/mirth-work rustc/coverage-run.sh @@ -16,7 +16,9 @@ name=$1; shift out=${COV_SUITES:-$work/cov-suites}/$name mkdir -p "$out/logs" rm -f "$out/done" -python3 "$here/coverage-compact.py" --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & +# The compactor is a mirth-lab subcommand: build it first (a no-op when up to date). +(cd "$here/.." && cargo build --release -q --offline -p mirth-lab) || exit 1 +"$here/../target/release/mirth-lab" coverage-compact --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & compactor=$! MIRTH_OUT=$out/logs "$@" > "$out/run.log" 2>&1 status=$? diff --git a/rustc/coverage-suites.sh b/rustc/coverage-suites.sh index 87e32bd..bef8686 100755 --- a/rustc/coverage-suites.sh +++ b/rustc/coverage-suites.sh @@ -2,12 +2,12 @@ # Run rustc's own test suites through compiletest with the coverage-instrumented compiler # (rustc/build.sh with MIRTH_WATCH=rustc/coverage.toml and BUILD_DIR), recording which of the # compiler's functions each rustc process reaches. Logs are folded as they finish -# (rustc/coverage-compact.py), so the disk holds only the union and what each test added. +# (`mirth-lab coverage-compact`), so the disk holds only the union and what each test added. # # MIRTH_RUST= BUILD_DIR= \ # rustc/coverage-suites.sh [-- ] # -# Writes //{union.txt,added.jsonl,x.log}. Read with rustc/coverage.py +# Writes //{union.txt,added.jsonl,x.log}. Read with `mirth-lab coverage` # --union /*/union.txt. set -uo pipefail : "${MIRTH_RUST:?set MIRTH_RUST}" @@ -27,7 +27,9 @@ rm -f "$out/done" export RUSTFLAGS_BOOTSTRAP="-L dependency=$BUILD_DIR/mirth-runtime" export RUSTFLAGS_NOT_BOOTSTRAP="$RUSTFLAGS_BOOTSTRAP" -python3 "$here/coverage-compact.py" --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & +# The compactor is a mirth-lab subcommand: build it first (a no-op when up to date). +(cd "$here/.." && cargo build --release -q --offline -p mirth-lab) || exit 1 +"$here/../target/release/mirth-lab" coverage-compact --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & compactor=$! # A test compile running more than five minutes is killed and named. From 4d64d8f80652ba888a0794d483fef92a09df3e67 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:02:49 +0000 Subject: [PATCH 17/23] Cargo.lock: mirth-lab's libc dependency Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- Cargo.lock | 1 + 1 file changed, 1 insertion(+) diff --git a/Cargo.lock b/Cargo.lock index 3851a8e..d985a42 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -321,6 +321,7 @@ version = "0.0.0" dependencies = [ "anyhow", "clap", + "libc", "mirth-rewrite", "rayon", "regex", From 16110ec246f977dc0272c4654c95cabdbefa480b Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:03:41 +0000 Subject: [PATCH 18/23] wip: fuzz, fuzz-replay, replay in mirth-lab (own mutations and cargo modules, to reconcile with port-ui) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- Cargo.lock | 185 +++++++- crates/mirth-lab/Cargo.toml | 4 + crates/mirth-lab/src/cargo.rs | 300 ++++++++++++ crates/mirth-lab/src/lib.rs | 2 + crates/mirth-lab/src/main.rs | 12 + crates/mirth-lab/src/mutations.rs | 426 +++++++++++++++++ crates/mirth-lab/src/tools/fuzz.rs | 533 ++++++++++++++++++++++ crates/mirth-lab/src/tools/fuzz_replay.rs | 118 +++++ crates/mirth-lab/src/tools/replay.rs | 314 +++++++++++++ 9 files changed, 1892 insertions(+), 2 deletions(-) create mode 100644 crates/mirth-lab/src/cargo.rs create mode 100644 crates/mirth-lab/src/mutations.rs create mode 100644 crates/mirth-lab/src/tools/fuzz.rs create mode 100644 crates/mirth-lab/src/tools/fuzz_replay.rs create mode 100644 crates/mirth-lab/src/tools/replay.rs diff --git a/Cargo.lock b/Cargo.lock index 3851a8e..2eb3cb9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2,6 +2,12 @@ # It is not intended for manual editing. version = 4 +[[package]] +name = "adler2" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" + [[package]] name = "aho-corasick" version = "1.1.5" @@ -151,6 +157,15 @@ dependencies = [ "libc", ] +[[package]] +name = "crc32fast" +version = "1.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78" +dependencies = [ + "cfg-if", +] + [[package]] name = "crossbeam-deque" version = "0.8.8" @@ -224,6 +239,27 @@ version = "2.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" +[[package]] +name = "filetime" +version = "0.2.29" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5c287a33c7f0a620c38e641e7f60827713987b3c0f26e8ddc9462cc69cf75759" +dependencies = [ + "cfg-if", + "libc", +] + +[[package]] +name = "flate2" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" +dependencies = [ + "crc32fast", + "miniz_oxide", + "zlib-rs", +] + [[package]] name = "generic-array" version = "0.14.7" @@ -234,6 +270,18 @@ dependencies = [ "version_check", ] +[[package]] +name = "getrandom" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" +dependencies = [ + "cfg-if", + "libc", + "r-efi 5.3.0", + "wasip2", +] + [[package]] name = "getrandom" version = "0.4.3" @@ -242,7 +290,7 @@ checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" dependencies = [ "cfg-if", "libc", - "r-efi", + "r-efi 6.0.0", ] [[package]] @@ -297,6 +345,16 @@ version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" +[[package]] +name = "miniz_oxide" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" +dependencies = [ + "adler2", + "simd-adler32", +] + [[package]] name = "mirth" version = "0.0.0" @@ -321,12 +379,17 @@ version = "0.0.0" dependencies = [ "anyhow", "clap", + "flate2", + "libc", "mirth-rewrite", + "rand", "rayon", "regex", "serde", "serde_json", "sha2", + "similar", + "tar", "tempfile", "wait-timeout", "walkdir", @@ -367,6 +430,15 @@ version = "1.70.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" +[[package]] +name = "ppv-lite86" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +dependencies = [ + "zerocopy", +] + [[package]] name = "proc-macro2" version = "1.0.107" @@ -385,12 +457,47 @@ dependencies = [ "proc-macro2", ] +[[package]] +name = "r-efi" +version = "5.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" + [[package]] name = "r-efi" version = "6.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" +[[package]] +name = "rand" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" +dependencies = [ + "rand_chacha", + "rand_core", +] + +[[package]] +name = "rand_chacha" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" +dependencies = [ + "ppv-lite86", + "rand_core", +] + +[[package]] +name = "rand_core" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" +dependencies = [ + "getrandom 0.3.4", +] + [[package]] name = "rayon" version = "1.12.0" @@ -525,6 +632,18 @@ dependencies = [ "digest", ] +[[package]] +name = "simd-adler32" +version = "0.3.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" + +[[package]] +name = "similar" +version = "2.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbbb5d9659141646ae647b42fe094daf6c6192d1620870b449d9557f748b2daa" + [[package]] name = "strsim" version = "0.11.1" @@ -553,6 +672,17 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "tar" +version = "0.4.46" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f6221d9a6003c78398e3b239969f352578258df48c8eb051caadae0015bc840" +dependencies = [ + "filetime", + "libc", + "xattr", +] + [[package]] name = "tempfile" version = "3.27.0" @@ -560,7 +690,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" dependencies = [ "fastrand", - "getrandom", + "getrandom 0.4.3", "once_cell", "rustix", "windows-sys", @@ -654,6 +784,15 @@ dependencies = [ "winapi-util", ] +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen", +] + [[package]] name = "winapi-util" version = "0.1.11" @@ -690,6 +829,48 @@ version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" + +[[package]] +name = "xattr" +version = "1.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32e45ad4206f6d2479085147f02bc2ef834ac85886624a23575ae137c8aa8156" +dependencies = [ + "libc", + "rustix", +] + +[[package]] +name = "zerocopy" +version = "0.8.61" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "879fb705ce98c32e41ebdb970fbe1204f8492423b314c6ab0354c3e7b5542866" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.61" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "708882a28301d604fa039cc7727607a98d04c4b86dc76ec9cc683f805d709759" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zlib-rs" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112" + [[package]] name = "zmij" version = "1.0.23" diff --git a/crates/mirth-lab/Cargo.toml b/crates/mirth-lab/Cargo.toml index 83fe69a..39918ee 100644 --- a/crates/mirth-lab/Cargo.toml +++ b/crates/mirth-lab/Cargo.toml @@ -22,4 +22,8 @@ tempfile = "3" walkdir = "2" wait-timeout = "0.2" libc = "0.2" +flate2 = "1" +rand = { version = "0.9", default-features = false, features = ["std", "std_rng"] } +similar = "2" +tar = "0.4" mirth-rewrite = { path = "../mirth-rewrite" } diff --git a/crates/mirth-lab/src/cargo.rs b/crates/mirth-lab/src/cargo.rs new file mode 100644 index 0000000..617a2f6 --- /dev/null +++ b/crates/mirth-lab/src/cargo.rs @@ -0,0 +1,300 @@ +//! Cargo builds two of which can be compared: what a build produced, read from the JSON messages +//! of `cargo build --message-format=json-render-diagnostics` for packages built from a path: +//! +//! rmeta every .rmeta, by path +//! rlib every .rlib's members by name, session suffixes removed (`artifacts`) +//! exe every executable +//! diag every diagnostic, rendered, counted per crate +//! +//! Also what the compiler's own checks print (RUSTC_VERIFY_REUSE, RUSTC_REPORT_UNTRACKED), and +//! the process-group timeout and tree copies the incremental fuzzers need. + +use std::collections::{BTreeMap, BTreeSet}; +use std::io::Read; +use std::os::unix::process::CommandExt; +use std::path::{Path, PathBuf}; +use std::process::{Command, Stdio}; +use std::sync::LazyLock; +use std::time::Duration; + +use regex::Regex; +use serde::Deserialize; +use wait_timeout::ChildExt; + +use crate::artifacts::{normalized_rlib, sha256}; + +#[derive(Deserialize)] +pub struct Message { + #[serde(default)] + pub reason: String, + #[serde(default)] + pub package_id: String, + #[serde(default)] + pub target: Option, + #[serde(default)] + pub filenames: Vec, + #[serde(default)] + pub executable: Option, + #[serde(default)] + pub fresh: bool, + #[serde(default)] + pub message: Option, +} + +#[derive(Deserialize)] +pub struct Target { + pub name: String, + #[serde(default)] + pub kind: Vec, +} + +impl Message { + pub fn target_name(&self) -> &str { + self.target.as_ref().map_or("", |t| t.name.as_str()) + } + pub fn from_path(&self) -> bool { + self.package_id.contains("path+file") + } +} + +/// Cargo's JSON messages in `stdout`, skipping lines that are not. +pub fn messages(stdout: &str) -> impl Iterator + '_ { + stdout.lines().filter_map(|l| serde_json::from_str(l).ok()) +} + +/// `f` relative to `target` when it is inside it. +pub fn relative(f: &str, target: &Path) -> String { + Path::new(f).strip_prefix(target).map_or_else(|_| f.to_owned(), |p| p.to_string_lossy().into_owned()) +} + +#[derive(Default, Clone, PartialEq)] +pub struct Collected { + pub rmeta: BTreeMap, + pub rlib: BTreeMap>, + pub exe: BTreeMap, + /// (crate, rendered diagnostic) -> count + pub diag: BTreeMap<(String, String), usize>, +} + +static SESSION: LazyLock = + LazyLock::new(|| regex::bytes::Regex::new(r"\.[0-9a-z]{7}(\.rcgu\.(?:o|dwo))").unwrap()); + +/// Artifacts from cargo's JSON messages, keyed by path relative to `target`. +pub fn collect(stdout: &str, target: &Path) -> Collected { + let mut found = Collected::default(); + for msg in messages(stdout) { + if !msg.from_path() { + continue; + } + if msg.reason == "compiler-message" { + let m = msg.message.as_ref(); + let text = m + .and_then(|m| m.get("rendered").and_then(|r| r.as_str()).filter(|s| !s.is_empty())) + .or_else(|| m.and_then(|m| m.get("message")).and_then(|r| r.as_str())) + .unwrap_or(""); + *found.diag.entry((msg.target_name().to_owned(), text.to_owned())).or_default() += 1; + continue; + } + if msg.reason != "compiler-artifact" { + continue; + } + for f in &msg.filenames { + let rel = relative(f, target); + if f.ends_with(".rmeta") { + found.rmeta.insert(rel, sha256(&std::fs::read(f).unwrap_or_default())); + } else if f.ends_with(".rlib") { + found.rlib.insert(rel, normalized_rlib(Path::new(f))); + } + } + if let Some(f) = &msg.executable { + let data = std::fs::read(f).unwrap_or_default(); + found.exe.insert(relative(f, target), sha256(&SESSION.replace_all(&data, &b"$1"[..]))); + } + } + found +} + +fn differing(a: &BTreeMap, b: &BTreeMap) -> Vec { + let keys: BTreeSet<&String> = a.keys().chain(b.keys()).collect(); + keys.into_iter().filter(|k| a.get(*k) != b.get(*k)).cloned().collect() +} + +/// {kind: [what differs]} for the kinds that differ between two collections. +pub fn compare(a: &Collected, b: &Collected) -> BTreeMap> { + let mut out = BTreeMap::new(); + for (kind, x, y) in [("rmeta", &a.rmeta, &b.rmeta), ("exe", &a.exe, &b.exe)] { + let d = differing(x, y); + if !d.is_empty() { + out.insert(kind.to_owned(), d); + } + } + let empty = BTreeMap::new(); + let mut rlib = Vec::new(); + for rel in a.rlib.keys().chain(b.rlib.keys()).collect::>() { + let members = differing(a.rlib.get(rel).unwrap_or(&empty), b.rlib.get(rel).unwrap_or(&empty)); + if !members.is_empty() { + let more = if members.len() > 5 { " …" } else { "" }; + rlib.push(format!("{rel}: {}{more}", members[..members.len().min(5)].join(", "))); + } + } + if !rlib.is_empty() { + out.insert("rlib".to_owned(), rlib); + } + if a.diag != b.diag { + let only = |x: &BTreeMap<(String, String), usize>, y: &BTreeMap<(String, String), usize>, which: &str| { + x.iter() + .filter(|(k, n)| y.get(*k).copied().unwrap_or(0) < **n) + .map(|((c, t), _)| format!("{c} only in the {which}: {:?}", t.chars().take(200).collect::())) + .collect::>() + }; + let mut d = only(&a.diag, &b.diag, "first"); + d.extend(only(&b.diag, &a.diag, "second")); + out.insert("diag".to_owned(), d); + } + out +} + +/// What the compiler's own check of reused results (docs/hunt/verify-reuse.patch) found stale: +/// `query `, `metadata`, `codegen unit` or `allocation sharing `, once each. +pub fn reuse_checks(log: &str) -> Vec { + static QUERY: LazyLock = LazyLock::new(|| Regex::new(r"query `(\w+)`").unwrap()); + static MODE: LazyLock = LazyLock::new(|| Regex::new(r"TypingModeEqWrapper\((\w+)\)").unwrap()); + let mut found = BTreeSet::new(); + for line in log.lines() { + if line.starts_with("rustc-verify-reuse: query `") { + found.insert(format!("query {}", line.split('`').nth(1).unwrap_or(""))); + } else if line.starts_with("rustc-verify-reuse: metadata") { + found.insert("metadata".to_owned()); + } else if line.starts_with("rustc-verify-reuse: codegen unit") { + found.insert("codegen unit".to_owned()); + } else if line.starts_with("rustc-verify-reuse: allocation shared differently") { + // Named by the two queries and typing modes, so that a new pattern is kept apart + // from a known one. + let mut parts: Vec = QUERY.captures_iter(line).map(|c| c[1].to_owned()).collect(); + let modes: BTreeSet = MODE.captures_iter(line).map(|c| c[1].to_owned()).collect(); + parts.extend(modes); + found.insert(format!("allocation sharing {}", parts.join(" / "))); + } + } + found.into_iter().collect() +} + +/// Reads of untracked state the compiler reported (docs/hunt/report-untracked.patch). +pub fn untracked_reads(log: &str) -> BTreeSet { + log.lines().filter(|l| l.starts_with("rustc-untracked-read:")).map(|l| l.trim().to_owned()).collect() +} + +/// Adds new reports of untracked reads to `path`, which lists each once. +pub fn note_untracked(path: &Path, lines: &BTreeSet) { + let known: BTreeSet = std::fs::read_to_string(path).unwrap_or_default().lines().map(str::to_owned).collect(); + let new: String = lines.difference(&known).map(|l| format!("{l}\n")).collect(); + if !new.is_empty() { + use std::io::Write; + if let Ok(mut f) = std::fs::OpenOptions::new().create(true).append(true).open(path) { + let _ = f.write_all(new.as_bytes()); + } + } +} + +/// Whether a compiler log shows a crash. +pub fn is_ice(log: &str) -> bool { + log.contains("internal compiler error") || log.contains("the compiler unexpectedly panicked") +} + +/// How a build command ended. +pub struct Run { + pub ok: bool, + pub stdout: String, + pub stderr: String, + /// Killed at the timeout while a rustc was still running. + pub hang: bool, +} + +/// Run `cmd` in its own process group; after `timeout` the group is killed, and the run counts +/// as a hang if a rustc was still running in it (a looping build script is the fixture's +/// problem, not the compiler's). +pub fn run_group(mut cmd: Command, timeout: Option) -> std::io::Result { + cmd.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()).process_group(0); + let mut child = cmd.spawn()?; + let pid = child.id() as i32; + let (mut out, mut err) = (child.stdout.take().expect("piped"), child.stderr.take().expect("piped")); + let out_thread = std::thread::spawn(move || { + let mut s = Vec::new(); + let _ = out.read_to_end(&mut s); + s + }); + let err_thread = std::thread::spawn(move || { + let mut s = Vec::new(); + let _ = err.read_to_end(&mut s); + s + }); + let status = match timeout { + Some(t) => child.wait_timeout(t)?, + None => Some(child.wait()?), + }; + let mut note = String::new(); + let (ok, hang) = match status { + Some(s) => (s.success(), false), + None => { + let ps = Command::new("ps").args(["-o", "args=", "-g", &pid.to_string()]).output(); + let group = ps.map(|o| String::from_utf8_lossy(&o.stdout).into_owned()).unwrap_or_default(); + let hang = group.lines().filter_map(|l| l.split_whitespace().next()).any(|a| a.ends_with("/rustc")); + // SAFETY: signalling the process group we created. + unsafe { libc::kill(-pid, libc::SIGKILL) }; + let _ = child.wait(); + note = format!("\nkilled after {}s; still running:\n{group}", timeout.unwrap().as_secs()); + (false, hang) + } + }; + let stdout = String::from_utf8_lossy(&out_thread.join().unwrap_or_default()).into_owned(); + let mut stderr = String::from_utf8_lossy(&err_thread.join().unwrap_or_default()).into_owned(); + stderr.push_str(¬e); + Ok(Run { ok, stdout, stderr, hang }) +} + +/// The last `n` characters of `s`. +pub fn tail(s: &str, n: usize) -> &str { + let count = s.chars().count(); + if count <= n { + return s; + } + let skip = s.char_indices().nth(count - n).map_or(0, |(i, _)| i); + &s[skip..] +} + +/// Copy a tree, skipping entries named in `ignore` at any depth, keeping modification times +/// (Cargo's freshness depends on them). Symbolic links are copied as links with `symlinks`, +/// followed otherwise. +pub fn copy_tree(from: &Path, to: &Path, ignore: &[&str], symlinks: bool) -> std::io::Result<()> { + let walk = walkdir::WalkDir::new(from).follow_links(!symlinks).into_iter().filter_entry(|e| { + e.depth() == 0 || !ignore.iter().any(|i| e.file_name() == *i) + }); + for e in walk { + let e = e.map_err(std::io::Error::other)?; + let dest: PathBuf = to.join(e.path().strip_prefix(from).unwrap()); + let ft = e.file_type(); + if ft.is_symlink() { + std::os::unix::fs::symlink(std::fs::read_link(e.path())?, &dest)?; + } else if ft.is_dir() { + std::fs::create_dir_all(&dest)?; + } else { + std::fs::copy(e.path(), &dest)?; + let meta = e.metadata().map_err(std::io::Error::other)?; + let times = std::fs::FileTimes::new().set_modified(meta.modified()?).set_accessed(meta.accessed()?); + std::fs::File::options().write(true).open(&dest)?.set_times(times)?; + } + } + Ok(()) +} + +/// Free bytes on the file system holding `path`. +pub fn free_bytes(path: &Path) -> u64 { + use std::os::unix::ffi::OsStrExt; + let Ok(c) = std::ffi::CString::new(path.as_os_str().as_bytes()) else { return u64::MAX }; + // SAFETY: statvfs is plain data, filled by the call for a valid C string. + let mut s: libc::statvfs = unsafe { std::mem::zeroed() }; + if unsafe { libc::statvfs(c.as_ptr(), &mut s) } != 0 { + return u64::MAX; + } + s.f_bavail as u64 * s.f_frsize as u64 +} diff --git a/crates/mirth-lab/src/lib.rs b/crates/mirth-lab/src/lib.rs index 0d8d916..7a3c245 100644 --- a/crates/mirth-lab/src/lib.rs +++ b/crates/mirth-lab/src/lib.rs @@ -2,8 +2,10 @@ //! `mirth-lab` binary, in `src/tools/`; docs/checks.md says what each looks for and found. pub mod artifacts; +pub mod cargo; pub mod driver; pub mod miri; +pub mod mutations; pub mod normalize; pub mod rustc; pub mod uitest; diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs index c4dbcea..6306481 100644 --- a/crates/mirth-lab/src/main.rs +++ b/crates/mirth-lab/src/main.rs @@ -10,11 +10,14 @@ mod tools { pub mod callgraph; pub mod crash_diff; pub mod diag_check; + pub mod fuzz; + pub mod fuzz_replay; pub mod gate_check; pub mod instr_check; pub mod miri_diff; pub mod opt_diff; pub mod release_diff; + pub mod replay; pub mod repro_diff; pub mod rewrite_diff; pub mod scale_check; @@ -62,6 +65,12 @@ enum Check { AbiDiff(tools::abi_diff::Args), /// Reachability over the compiler's call graph; coverage of what can run, and gap lists. Callgraph(tools::callgraph::Args), + /// Random edits to a fixture; each incremental rebuild must match a clean build (P6). + Fuzz(tools::fuzz::Args), + /// Replay a fuzz finding's edits and compare the last incremental build with a clean one. + FuzzReplay(tools::fuzz_replay::Args), + /// A crate's git history through incremental builds, each compared with a clean build. + Replay(tools::replay::Args), } fn main() -> ExitCode { @@ -82,6 +91,9 @@ fn main() -> ExitCode { Check::ScaleCheck(a) => tools::scale_check::run(a), Check::AbiDiff(a) => tools::abi_diff::run(a), Check::Callgraph(a) => tools::callgraph::run(a), + Check::Fuzz(a) => tools::fuzz::run(a), + Check::FuzzReplay(a) => tools::fuzz_replay::run(a), + Check::Replay(a) => tools::replay::run(a), }; match result { Ok(code) => code, diff --git a/crates/mirth-lab/src/mutations.rs b/crates/mirth-lab/src/mutations.rs new file mode 100644 index 0000000..489e738 --- /dev/null +++ b/crates/mirth-lab/src/mutations.rs @@ -0,0 +1,426 @@ +//! Random mechanical edits to Rust source, shared by the fuzzers. +//! +//! Each edit takes the file's text, a random source and a counter (which names what the edit +//! adds), and returns the new text, or None when it does not apply. [`EDITS`] lists them with +//! weights; [`pick`] draws one. + +use std::sync::LazyLock; + +use regex::Regex; + +/// The random choices an edit makes. Implemented for every `rand::Rng`; the draws mirror +/// Python's `random` (randrange, choice, choices, shuffle) so that a scripted source replays +/// the same edit in both. +pub trait EditRng { + /// A float in [0, 1). + fn float(&mut self) -> f64; + /// An integer in [0, n), n > 0. + fn below(&mut self, n: usize) -> usize; +} + +impl EditRng for R { + fn float(&mut self) -> f64 { + self.random::() + } + fn below(&mut self, n: usize) -> usize { + self.random_range(0..n) + } +} + +fn choice<'a, T>(rng: &mut dyn EditRng, v: &'a [T]) -> &'a T { + &v[rng.below(v.len())] +} + +/// Fisher-Yates from the end, as Python's `random.shuffle`. +fn shuffle(rng: &mut dyn EditRng, v: &mut [T]) { + for i in (1..v.len()).rev() { + let j = rng.below(i + 1); + v.swap(i, j); + } +} + +pub type EditFn = fn(&str, &mut dyn EditRng, u64) -> Option; + +pub struct Edit { + pub name: &'static str, + pub weight: u32, + pub apply: EditFn, +} + +pub const EDITS: &[Edit] = &[ + Edit { name: "comment_line", weight: 10, apply: comment_line }, + Edit { name: "blank_line", weight: 6, apply: blank_line }, + Edit { name: "remove_comment", weight: 3, apply: remove_comment }, + Edit { name: "indent_line", weight: 4, apply: indent_line }, + Edit { name: "swap_items", weight: 6, apply: swap_items }, + Edit { name: "move_item_to_end", weight: 3, apply: move_item_to_end }, + Edit { name: "delete_item", weight: 2, apply: delete_item }, + Edit { name: "duplicate_fn", weight: 4, apply: duplicate_fn }, + Edit { name: "add_item", weight: 8, apply: add_item }, + Edit { name: "int_literal", weight: 6, apply: int_literal }, + Edit { name: "str_literal", weight: 4, apply: str_literal }, + Edit { name: "toggle_inline", weight: 4, apply: toggle_inline }, + Edit { name: "doc_comment", weight: 4, apply: doc_comment }, + Edit { name: "reorder_derive", weight: 2, apply: reorder_derive }, + Edit { name: "narrow_visibility", weight: 2, apply: narrow_visibility }, + Edit { name: "rename_local", weight: 3, apply: rename_local }, +]; + +/// An edit drawn by weight (Python's `random.choices`). +pub fn pick(rng: &mut dyn EditRng) -> &'static Edit { + let total: u32 = EDITS.iter().map(|e| e.weight).sum(); + let u = rng.float() * total as f64; + let mut acc = 0.0; + for e in EDITS { + acc += e.weight as f64; + if u < acc { + return e; + } + } + &EDITS[EDITS.len() - 1] +} + +/// The edits that change literals: not for build scripts (Cargo keeps stale OUT_DIR files, so +/// renaming a generated file splits the builds, and a changed number can make it loop). +pub fn changes_literals(e: &Edit) -> bool { + e.name == "int_literal" || e.name == "str_literal" +} + +/// Top-level items: blank-line separated, continuation blocks merged. +fn blocks(text: &str) -> Vec { + let mut out: Vec = Vec::new(); + for b in text.split("\n\n") { + let continues = b.chars().next().is_some_and(char::is_whitespace) || b.starts_with('}') || b.starts_with("where"); + match out.last_mut() { + Some(last) if continues => { + last.push_str("\n\n"); + last.push_str(b); + } + _ => out.push(b.to_owned()), + } + } + out +} + +fn lines(text: &str) -> Vec { + text.split('\n').map(str::to_owned).collect() +} + +fn indent_of(line: &str) -> &str { + &line[..line.len() - line.trim_start().len()] +} + +fn comment_line(text: &str, rng: &mut dyn EditRng, n: u64) -> Option { + let mut l = lines(text); + let i = rng.below(l.len() + 1); + let indent = l.get(i).map_or("", |s| indent_of(s)).to_owned(); + l.insert(i, format!("{indent}// fuzz {n}")); + Some(l.join("\n")) +} + +fn blank_line(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let mut l = lines(text); + let i = rng.below(l.len() + 1); + l.insert(i, String::new()); + Some(l.join("\n")) +} + +fn remove_comment(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let mut l = lines(text); + let idx: Vec = + (0..l.len()).filter(|&i| l[i].trim().starts_with("//") && !l[i].trim().starts_with("//!")).collect(); + if idx.is_empty() { + return None; + } + l.remove(*choice(rng, &idx)); + Some(l.join("\n")) +} + +fn indent_line(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let mut l = lines(text); + let idx: Vec = (0..l.len()).filter(|&i| !l[i].trim().is_empty()).collect(); + if idx.is_empty() { + return None; + } + let i = *choice(rng, &idx); + l[i] = format!(" {}", l[i]); + Some(l.join("\n")) +} + +fn swap_items(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + let i = 1 + rng.below(b.len() - 2); + b.swap(i, i + 1); + Some(b.join("\n\n")) +} + +fn move_item_to_end(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + let i = 1 + rng.below(b.len() - 1); + let item = b.remove(i); + b.push(item.trim_end_matches('\n').to_owned()); + Some(b.join("\n\n") + "\n") +} + +fn delete_item(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + b.remove(1 + rng.below(b.len() - 1)); + Some(b.join("\n\n")) +} + +static FN_ITEM: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)^(pub(\([^)]*\))? )?(const )?(async )?fn \w+").unwrap()); +static FN_NAME: LazyLock = LazyLock::new(|| Regex::new(r"\bfn (\w+)").unwrap()); + +fn duplicate_fn(text: &str, rng: &mut dyn EditRng, n: u64) -> Option { + let mut b = blocks(text); + let fns: Vec = (0..b.len()).filter(|&i| FN_ITEM.is_match(&b[i])).collect(); + if fns.is_empty() { + return None; + } + let i = *choice(rng, &fns); + let copy = FN_NAME.replacen(&b[i], 1, |c: ®ex::Captures| format!("fn {}_fuzz{n}", &c[1])).into_owned(); + b.insert(i + 1, copy); + Some(b.join("\n\n")) +} + +const ADDITIONS: &[&str] = &[ + "fn fuzz_private_{n}() -> u32 { {n} }", + "pub fn fuzz_public_{n}(x: u32) -> u32 { x.wrapping_mul({n}) }", + "#[inline]\npub fn fuzz_inline_{n}(x: &T) -> (T, u32) { (x.clone(), {n}) }", + "pub const FUZZ_{n}: &str = \"fuzz {n}\";", + "pub static FUZZ_STATIC_{n}: [u8; 3] = [{n} as u8, 1, 2];", + "#[derive(Debug, Clone, PartialEq)]\npub struct Fuzz{n} { pub items: [T; N], pub tag: &'static str }", + "pub enum FuzzEnum{n} { A(u32), B { x: i64 }, C }", + "pub trait FuzzTrait{n} { fn go(&self) -> impl Sized; const K: u32 = {n}; }", + "pub async fn fuzz_async_{n}() -> u32 { {n} }", + "pub type FuzzAlias{n} = Vec<(T, u32)>;", + "macro_rules! fuzz_macro_{n} { ($e:expr) => { $e + {n} }; }", + "pub mod fuzz_mod_{n} { pub fn inner() -> &'static str { \"{n}\" } }", +]; + +fn add_item(text: &str, rng: &mut dyn EditRng, n: u64) -> Option { + let mut b = blocks(text); + let i = 1 + rng.below(b.len()); + b.insert(i, choice(rng, ADDITIONS).replace("{n}", &n.to_string())); + Some(b.join("\n\n")) +} + +/// Python's `\w`: alphanumeric or underscore. +fn word(c: char) -> bool { + c.is_alphanumeric() || c == '_' +} + +static DIGITS: LazyLock = LazyLock::new(|| Regex::new(r"\d+").unwrap()); + +/// One more than a decimal number written in ASCII or other digits, as Python's `int(s) + 1`. +fn increment(s: &str) -> Option { + let mut digits: Vec = s.chars().map(|c| c.to_digit(10).map(|d| d as u8)).collect::>()?; + let mut i = digits.len(); + loop { + if i == 0 { + digits.insert(0, 1); + break; + } + i -= 1; + if digits[i] == 9 { + digits[i] = 0; + } else { + digits[i] += 1; + break; + } + } + let first = digits.iter().position(|&d| d != 0).unwrap_or(digits.len() - 1); + Some(digits[first..].iter().map(|d| (b'0' + d) as char).collect()) +} + +/// Numbers not part of a word or a float: `(? Option { + let ms: Vec = DIGITS + .find_iter(text) + .filter(|m| { + let before = text[..m.start()].chars().next_back(); + let after = text[m.end()..].chars().next(); + !before.is_some_and(|c| word(c) || c == '.') && !after.is_some_and(|c| word(c) || c == '.') + }) + .collect(); + if ms.is_empty() { + return None; + } + let m = choice(rng, &ms); + Some(format!("{}{}{}", &text[..m.start()], increment(m.as_str())?, &text[m.end()..])) +} + +/// Plain string literals: `(? Vec<(usize, usize)> { + let mut out = Vec::new(); + let mut i = 0; + while let Some(off) = text[i..].find('"') { + let start = i + off; + let before = text[..start].chars().next_back(); + if !before.is_some_and(|c| word(c) || c == '\\') { + let body = &text[start + 1..]; + let stop = body.find(['"', '\\', '\n']); + if let Some(s) = stop + && body.as_bytes()[s] == b'"' + { + let end = start + 1 + s + 1; + out.push((start, end)); + i = end; + continue; + } + } + i = start + 1; + } + out +} + +fn str_literal(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let ms = str_literals(text); + if ms.is_empty() { + return None; + } + let (s, e) = *choice(rng, &ms); + Some(format!("{}\"{}~\"{}", &text[..s], &text[s + 1..e - 1], &text[e..])) +} + +static FN_LINE: LazyLock = LazyLock::new(|| Regex::new(r"^\s*(pub(\([^)]*\))? )?(const )?fn ").unwrap()); + +fn toggle_inline(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let mut l = lines(text); + let inl: Vec = + (0..l.len()).filter(|&i| matches!(l[i].trim(), "#[inline]" | "#[inline(never)]" | "#[inline(always)]")).collect(); + let fns: Vec = (0..l.len()).filter(|&i| FN_LINE.is_match(&l[i])).collect(); + if !inl.is_empty() && rng.float() < 0.5 { + l.remove(*choice(rng, &inl)); + } else if !fns.is_empty() { + let i = *choice(rng, &fns); + let indent = indent_of(&l[i]).to_owned(); + let attr = choice(rng, &["#[inline]", "#[inline(never)]", "#[cold]", "#[must_use]"]); + l.insert(i, format!("{indent}{attr}")); + } else { + return None; + } + Some(l.join("\n")) +} + +static ITEM_LINE: LazyLock = + LazyLock::new(|| Regex::new(r"^\s*(pub(\([^)]*\))? )?(fn|struct|enum|trait|const|static|type|mod) ").unwrap()); + +fn doc_comment(text: &str, rng: &mut dyn EditRng, n: u64) -> Option { + let mut l = lines(text); + let idx: Vec = (0..l.len()).filter(|&i| ITEM_LINE.is_match(&l[i])).collect(); + if idx.is_empty() { + return None; + } + let i = *choice(rng, &idx); + let indent = indent_of(&l[i]).to_owned(); + l.insert(i, format!("{indent}/// Fuzz doc {n}, see [`Vec`].")); + Some(l.join("\n")) +} + +static DERIVE: LazyLock = LazyLock::new(|| Regex::new(r"#\[derive\(([^)]*)\)\]").unwrap()); + +fn reorder_derive(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let ms: Vec = DERIVE.captures_iter(text).collect(); + if ms.is_empty() { + return None; + } + let m = choice(rng, &ms); + let mut names: Vec<&str> = m[1].split(',').map(str::trim).filter(|x| !x.is_empty()).collect(); + if names.len() < 2 { + return None; + } + shuffle(rng, &mut names); + let all = m.get(0).unwrap(); + Some(format!("{}#[derive({})]{}", &text[..all.start()], names.join(", "), &text[all.end()..])) +} + +static PUB_ITEM: LazyLock = + LazyLock::new(|| Regex::new(r"\bpub (fn|struct|enum|const|static|trait|mod|type) ").unwrap()); + +fn narrow_visibility(text: &str, rng: &mut dyn EditRng, _: u64) -> Option { + let ms: Vec = PUB_ITEM.captures_iter(text).collect(); + if ms.is_empty() { + return None; + } + let m = choice(rng, &ms); + let all = m.get(0).unwrap(); + Some(format!("{}pub(crate) {} {}", &text[..all.start()], &m[1], &text[all.end()..])) +} + +static LET: LazyLock = LazyLock::new(|| Regex::new(r"\blet (mut )?([a-z_][a-z0-9_]*)\b").unwrap()); + +fn rename_local(text: &str, rng: &mut dyn EditRng, n: u64) -> Option { + let ms: Vec = LET.captures_iter(text).collect(); + if ms.is_empty() { + return None; + } + let m = choice(rng, &ms); + let name = &m[2]; + if name == "_" { + return None; + } + let (start, after) = (m.get(0).unwrap().start(), m.get(0).unwrap().end()); + // Rename from the binding to the end of the enclosing top-level block. + let end = text[after..].find("\n}\n").map_or(text.len(), |e| after + e); + let re = Regex::new(&format!(r"\b{name}\b")).unwrap(); + let body = re.replace_all(&text[start..end], format!("{name}_f{n}").as_str()); + Some(format!("{}{}{}", &text[..start], body, &text[end..])) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Draws from a fixed list of floats in [0, 1); `below(n)` is `floor(u * n)`. + pub struct Scripted(pub Vec, pub usize); + impl EditRng for Scripted { + fn float(&mut self) -> f64 { + let u = self.0[self.1 % self.0.len()]; + self.1 += 1; + u + } + fn below(&mut self, n: usize) -> usize { + ((self.float() * n as f64) as usize).min(n - 1) + } + } + + #[test] + fn literals() { + assert_eq!(increment("009").as_deref(), Some("10")); + assert_eq!(increment("99").as_deref(), Some("100")); + let mut r = Scripted(vec![0.0], 0); + assert_eq!(int_literal("a1 + 2.0 + 3", &mut r, 0).as_deref(), Some("a1 + 2.0 + 4")); + assert_eq!(str_literal(r#"r"x" + "y""#, &mut r, 0).as_deref(), Some(r#"r"x" + "y~""#)); + } + + /// Cases written by the Python edits with the same scripted draws (MIRTH_MUTATION_CASES: + /// a JSON list of {edit, text, n, draws, out}). + #[test] + fn same_as_python() { + let Ok(path) = std::env::var("MIRTH_MUTATION_CASES") else { return }; + let cases: Vec = serde_json::from_str(&std::fs::read_to_string(path).unwrap()).unwrap(); + let mut bad = 0; + for c in &cases { + let edit = EDITS.iter().find(|e| e.name == c["edit"]).unwrap(); + let draws: Vec = c["draws"].as_array().unwrap().iter().map(|v| v.as_f64().unwrap()).collect(); + let got = (edit.apply)(c["text"].as_str().unwrap(), &mut Scripted(draws, 0), c["n"].as_u64().unwrap()); + if got.as_deref() != c["out"].as_str() { + bad += 1; + eprintln!("{}: differs", c["edit"]); + } + } + assert_eq!(bad, 0, "of {} cases", cases.len()); + eprintln!("{} cases agree", cases.len()); + } +} diff --git a/crates/mirth-lab/src/tools/fuzz.rs b/crates/mirth-lab/src/tools/fuzz.rs new file mode 100644 index 0000000..c479633 --- /dev/null +++ b/crates/mirth-lab/src/tools/fuzz.rs @@ -0,0 +1,533 @@ +//! Make random edits to a fixture and check each incremental rebuild. +//! +//! Each worker keeps one copy of the fixture and repeats: +//! +//! 1. apply a random mechanical edit (`mutations`: a comment, a moved item, a changed +//! literal, a new function, ...); +//! 2. rebuild incrementally; if the edit does not compile, revert it (the next build then also +//! exercises recovery from a failed session); +//! 3. build the same source from scratch, at the same path; +//! 4. compare: +//! P6 every .rmeta Cargo reports for a workspace member +//! rlib every rlib's members, object code included +//! exe the binary's bytes +//! diag the diagnostics each crate printed +//! run the binaries' output and exit status +//! reuse the compiler's own check of what it reused (RUSTC_VERIFY_REUSE, +//! docs/hunt/verify-reuse.patch) found nothing stale +//! P5 when anything differs, clean builds are made again; if two clean builds +//! differ, that is reported instead (nondeterminism) +//! ICE neither build crashed the compiler +//! split both builds succeed or both fail +//! +//! Every --reset kept edits the worker starts again from the pristine fixture. A finding keeps +//! the edits since the last reset (history.json, which `fuzz-replay` replays), and both builds' +//! differing files and logs. +//! +//! Stop it early by creating /STOP. Progress is in /stats-w.json. With +//! --pause-on-finding all workers stop at the first finding (/PAUSED says which); patch +//! rustc and run again with the patched compiler. +//! +//! Seeds name the same edit sequence only within this implementation: the Python fuzzer drew +//! from Python's generator. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::{Duration, Instant}; + +use rand::SeedableRng; +use rand::rngs::StdRng; +use regex::Regex; +use serde_json::{Value, json}; + +use mirth_lab::cargo::{self, Collected, copy_tree, messages, relative, tail}; +use mirth_lab::mutations; + +#[derive(clap::Args, Debug, Clone)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + fixture: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + workers: u64, + /// Per worker. + #[arg(long, default_value_t = 1_000_000_000)] + edits: u64, + #[arg(long, default_value_t = 40)] + reset: usize, + /// Findings kept per kind. + #[arg(long, default_value_t = 5)] + keep: usize, + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "-Zincremental-verify-ich", allow_hyphen_values = true)] + rustflags: String, + #[arg(long, default_value_t = 0)] + seed: u64, + /// Seconds before a build counts as hung. + #[arg(long, default_value_t = 180)] + timeout: u64, + /// Clean builds made again when anything differs, to tell nondeterminism from P6. + #[arg(long, default_value_t = 12)] + p5_builds: usize, + /// Stop when the disk has less free. + #[arg(long, default_value_t = 20)] + min_free_gb: u64, + /// cargo check instead of cargo build: metadata only, no code or binaries. + #[arg(long)] + check: bool, + /// Do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch). + #[arg(long)] + no_verify_reuse: bool, + /// Pass --target to cargo, so RUSTFLAGS skip build scripts and proc macros. + #[arg(long)] + target: Option, + /// Stop all workers at the first finding (writes /PAUSED); patch rustc and run again. + #[arg(long)] + pause_on_finding: bool, +} + +struct Ctx { + args: Args, + work: PathBuf, + fixture: PathBuf, + bin: String, + /// Kinds not compared: metadata is compared separately; with split debuginfo, objects and + /// binaries name .dwo files by session, so two clean builds differ too. + skip: Vec<&'static str>, +} + +struct Build { + ok: bool, + ice: bool, + hang: bool, + reuse: Vec, + log: String, + rmetas: BTreeMap>, + exe: Option, + art: Collected, +} + +impl Ctx { + fn build(&self, src: &Path, target: &Path) -> Build { + let a = &self.args; + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{}", a.toolchain)) + .args([if a.check { "check" } else { "build" }, "--workspace", "--offline", "-j", "4", "--target-dir"]) + .arg(target) + .arg("--message-format=json-render-diagnostics") + .current_dir(src) + .env("RUSTC", &a.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + .env("RUSTFLAGS", &a.rustflags) + .env("CARGO_TERM_COLOR", "never"); + if let Some(t) = &a.target { + cmd.args(["--target", t]); + } + if !a.no_verify_reuse { + cmd.env("RUSTC_VERIFY_REUSE", "1").env("RUSTC_REPORT_UNTRACKED", "1"); + } + let r = cargo::run_group(cmd, Some(Duration::from_secs(a.timeout))).unwrap_or_else(|e| cargo::Run { + ok: false, + stdout: String::new(), + stderr: format!("cargo did not start: {e}"), + hang: false, + }); + let (mut rmetas, mut exe) = (BTreeMap::new(), None); + for msg in messages(&r.stdout) { + if msg.reason != "compiler-artifact" { + continue; + } + for f in &msg.filenames { + if f.ends_with(".rmeta") { + rmetas.insert(relative(f, target), std::fs::read(f).unwrap_or_default()); + } else if f.ends_with(".so") && msg.target.as_ref().is_some_and(|t| t.kind.iter().any(|k| k == "proc-macro")) { + // A proc macro's metadata is in the .rustc section of its shared library. + let section = Command::new("objcopy") + .args(["--dump-section", ".rustc=/dev/stdout", f, "/dev/null"]) + .output() + .map(|o| o.stdout) + .unwrap_or_default(); + rmetas.insert(relative(f, target) + ":.rustc", section); + } + } + if let Some(e) = &msg.executable + && msg.target_name() == self.bin + { + exe = Some(e.clone()); + } + } + let log = r.stderr; + Build { + ok: r.ok && !r.hang, + ice: cargo::is_ice(&log), + hang: r.hang, + reuse: cargo::reuse_checks(&log), + art: cargo::collect(&r.stdout, target), + rmetas, + exe, + log, + } + } +} + +/// Reports seen since the reuse check exists and judged benign (docs/shadow-mode.md): reused +/// codegen units differing only in debuginfo at the end of the file, and constant allocations +/// shared differently between evaluations in different typing modes. +fn known(kind: &str, detail: &[String]) -> bool { + kind == "verify-reuse" + && detail.iter().all(|d| d == "codegen unit" || d.starts_with("allocation sharing eval_to_const_value_raw")) +} + +fn run_exe(exe: Option<&str>) -> Value { + let Some(exe) = exe else { return Value::Null }; + match mirth_lab::rustc::run_command(Command::new(exe), Duration::from_secs(30)) { + Ok(f) => match f.exit { + mirth_lab::rustc::Exit::Timeout => json!(["timeout", ""]), + mirth_lab::rustc::Exit::Code(c) => json!([c, tail(&f.stdout_text(), 2000)]), + // Python's returncode for a signal is its negation. + mirth_lab::rustc::Exit::Signal(s) => json!([-s, tail(&f.stdout_text(), 2000)]), + }, + Err(_) => Value::Null, + } +} + +fn unified_diff(old: &str, new: &str, rel: &str) -> String { + similar::TextDiff::from_lines(old, new) + .unified_diff() + .header(&format!("a/{rel}"), &format!("b/{rel}")) + .to_string() +} + +fn rs_files(src: &Path) -> Vec { + let mut v: Vec = walkdir::WalkDir::new(src) + .into_iter() + .filter_entry(|e| e.depth() == 0 || e.file_name() != "target") + .filter_map(Result::ok) + .filter(|e| e.file_type().is_file() && e.path().extension().is_some_and(|x| x == "rs")) + .map(|e| e.into_path()) + .collect(); + v.sort(); + v +} + +fn tar_gz(src: &Path, out: &Path) -> std::io::Result<()> { + let f = std::fs::File::create(out)?; + let mut t = tar::Builder::new(flate2::write::GzEncoder::new(f, flate2::Compression::default())); + t.follow_symlinks(false); + t.append_dir_all(".", src)?; + t.into_inner()?.finish()?; + Ok(()) +} + +fn rmtree(p: &Path) { + let _ = std::fs::remove_dir_all(p); +} + +struct Worker<'a> { + ctx: &'a Ctx, + k: u64, + home: PathBuf, + src: PathBuf, + target: PathBuf, + inc_target: PathBuf, + findings: PathBuf, + stats: Stats, + kept: BTreeMap, + history: Vec, + previous: BTreeMap>, +} + +#[derive(serde::Serialize, Default)] +struct Stats { + edits: u64, + built: u64, + failed: u64, + compared: u64, + findings: BTreeMap, + secs: f64, + by_edit: BTreeMap, +} + +impl Worker<'_> { + fn reset(&mut self) -> bool { + rmtree(&self.home); + let _ = std::fs::create_dir_all(&self.home); + if let Err(e) = copy_tree(&self.ctx.fixture, &self.src, &["target", "edits", "edit"], false) { + let _ = std::fs::write(self.ctx.work.join(format!("error-w{}.log", self.k)), e.to_string()); + return false; + } + self.history.clear(); + let b = self.ctx.build(&self.src, &self.target); + self.previous = b.rmetas; + if !b.ok { + let _ = std::fs::write(self.ctx.work.join(format!("error-w{}.log", self.k)), tail(&b.log, 6000)); + return false; + } + true + } + + fn report(&mut self, kind: &str, detail: &[String], inc: &Build, clean: Option<&Build>, extra: Value) { + *self.stats.findings.entry(kind.to_owned()).or_default() += 1; + let work = &self.ctx.work; + if self.ctx.args.pause_on_finding && !known(kind, detail) { + let paused = json!({"worker": self.k, "edit": self.stats.edits, "kind": kind, "detail": detail}); + let _ = std::fs::write(work.join("PAUSED"), serde_json::to_string_pretty(&paused).unwrap()); + let _ = std::fs::write(work.join("STOP"), ""); + } + let mut sorted = detail.to_vec(); + sorted.sort(); + let key = format!("{kind}:{}", sorted.join(",")); + let n = self.kept.entry(key).or_default(); + if *n >= self.ctx.args.keep { + return; + } + *n += 1; + let d = self.findings.join(format!("w{}-{:07}-{kind}", self.k, self.stats.edits)); + let _ = std::fs::create_dir_all(&d); + let _ = std::fs::write(d.join("history.json"), serde_json::to_string_pretty(&self.history).unwrap()); + let finding = json!({"kind": kind, "detail": detail, "extra": extra}); + let _ = std::fs::write(d.join("finding.json"), serde_json::to_string_pretty(&finding).unwrap()); + let _ = std::fs::write(d.join("inc.log"), tail(&inc.log, 8000)); + for (name, b) in [("inc", Some(inc)), ("clean", clean)] { + let lines: Vec<&str> = b.map(|b| b.log.lines().collect()).unwrap_or_default(); + let hit = lines.iter().position(|l| { + l.contains("panicked at") + || l.contains("internal compiler error") + || l.contains("unexpectedly panicked") + || l.contains("interrupted by SIG") + }); + if let Some(h) = hit { + let excerpt = lines[h.saturating_sub(5)..(h + 60).min(lines.len())].join("\n"); + let _ = std::fs::write(d.join(format!("{name}-ice.txt")), excerpt); + } + } + let _ = std::fs::write(d.join("clean.log"), clean.map_or("", |c| tail(&c.log, 8000))); + for rel in detail { + let name = Path::new(rel).file_name().map(|s| s.to_string_lossy().into_owned()).unwrap_or_default(); + if let Some(v) = inc.rmetas.get(rel) { + let _ = std::fs::write(d.join(format!("{name}.inc")), v); + } + if let Some(v) = clean.and_then(|c| c.rmetas.get(rel)) { + let _ = std::fs::write(d.join(format!("{name}.clean")), v); + } + } + let _ = tar_gz(&self.src, &d.join("src.tar.gz")); + } + + fn write_stats(&self) { + let _ = std::fs::write( + self.ctx.work.join(format!("stats-w{}.json", self.k)), + serde_json::to_string(&self.stats).unwrap(), + ); + } + + fn run(&mut self) { + let _ = std::fs::create_dir_all(&self.findings); + if !self.reset() { + return; + } + let a = &self.ctx.args; + let mut rng = StdRng::seed_from_u64(a.seed * 1000 + self.k); + let started = Instant::now(); + let stop = self.ctx.work.join("STOP"); + while self.stats.edits < a.edits && !stop.exists() { + if cargo::free_bytes(&self.ctx.work) < a.min_free_gb << 30 { + println!("worker {}: less than {} GB free, stopping", self.k, a.min_free_gb); + break; + } + if self.history.iter().filter(|h| h["kept"] == true).count() >= a.reset && !self.reset() { + break; + } + let n = self.stats.edits; + self.stats.edits += 1; + let paths = rs_files(&self.src); + if paths.is_empty() { + break; + } + let path = paths[mutations::EditRng::below(&mut rng, paths.len())].clone(); + let old = std::fs::read_to_string(&path).unwrap_or_default(); + let edit = mutations::pick(&mut rng); + if mutations::changes_literals(edit) && path.file_name().is_some_and(|f| f == "build.rs") { + continue; + } + let new = match (edit.apply)(&old, &mut rng, n) { + Some(new) if new != old => new, + _ => continue, + }; + let _ = std::fs::write(&path, &new); + let rel = path.strip_prefix(&self.src).unwrap().to_string_lossy().into_owned(); + let diff = unified_diff(&old, &new, &rel); + let inc = self.ctx.build(&self.src, &self.target); + cargo::note_untracked(&self.ctx.work.join("untracked.txt"), &cargo::untracked_reads(&inc.log)); + self.history.push(json!({"edit": edit.name, "file": rel, "diff": diff, "before": old, "after": new, "kept": inc.ok})); + self.stats.by_edit.entry(edit.name.to_owned()).or_default()[0] += 1; + if inc.ice { + self.report("ICE", &[], &inc, None, Value::Null); + } + if inc.hang { + self.report("hang", &[], &inc, None, Value::Null); + } + if !inc.reuse.is_empty() { + let lines: Vec = inc + .log + .lines() + .filter(|l| l.starts_with("rustc-verify-reuse")) + .take(20) + .map(|l| l.chars().take(3000).collect()) + .collect(); + let reuse = inc.reuse.clone(); + self.report("verify-reuse", &reuse, &inc, None, json!(lines)); + } + if !inc.ok { + self.stats.failed += 1; + let _ = std::fs::write(&path, &old); + self.history.push(json!({"edit": "revert", "file": rel, "diff": "", "kept": false})); + continue; + } + self.stats.by_edit.get_mut(edit.name).unwrap()[1] += 1; + self.stats.built += 1; + self.compare(inc); + if self.stats.edits % 20 == 0 { + self.stats.secs = started.elapsed().as_secs_f64(); + self.write_stats(); + } + } + self.stats.secs = started.elapsed().as_secs_f64(); + self.write_stats(); + } + + /// The clean build at the same path, and the comparisons. + fn compare(&mut self, inc: Build) { + let (target, inc_target) = (self.target.clone(), self.inc_target.clone()); + rmtree(&inc_target); + let _ = std::fs::rename(&target, &inc_target); + let clean = self.ctx.build(&self.src, &target); + if clean.ice { + self.report("ICE", &["clean".into()], &inc, Some(&clean), Value::Null); + } + if clean.hang { + self.report("hang", &["clean".into()], &inc, Some(&clean), Value::Null); + } + if !clean.ok { + self.report("split", &[], &inc, Some(&clean), Value::Null); + } else { + self.stats.compared += 1; + let skip = &self.ctx.skip; + let rmeta_differ = |a: &BTreeMap>, b: &BTreeMap>| -> Vec { + let keys: std::collections::BTreeSet<&String> = a.keys().chain(b.keys()).collect(); + keys.into_iter().filter(|r| a.get(*r) != b.get(*r)).cloned().collect() + }; + let mut differ = rmeta_differ(&inc.rmetas, &clean.rmetas); + let mut others: BTreeMap> = + cargo::compare(&inc.art, &clean.art).into_iter().filter(|(k, _)| !skip.contains(&k.as_str())).collect(); + if !differ.is_empty() || !others.is_empty() { + // Build clean again: if two clean builds differ, the difference is + // nondeterminism (P5), not incremental reuse. + let mut p5: Vec = Vec::new(); + let mut again_ok = false; + let mut again = None; + for _ in 0..self.ctx.args.p5_builds { + rmtree(&target); + let b = self.ctx.build(&self.src, &target); + p5 = rmeta_differ(&clean.rmetas, &b.rmetas); + p5.extend( + cargo::compare(&clean.art, &b.art) + .into_iter() + .filter(|(k, _)| !skip.contains(&k.as_str())) + .map(|(k, v)| format!("{k}: {}", v[0])), + ); + again_ok = b.ok; + again = Some(b); + if !p5.is_empty() || !again_ok { + break; + } + } + if again_ok && !p5.is_empty() { + p5.truncate(10); + self.report("P5", &p5, &clean, again.as_ref(), Value::Null); + differ.clear(); + others.clear(); + } + } + // Known: metadata reused unchanged from the previous session although a source file + // changed (its hash and length in the source map are stale). + let stale = differ.iter().any(|r| self.previous.get(r).is_some_and(|p| Some(p) == inc.rmetas.get(r))); + if !differ.is_empty() && stale { + *self.stats.findings.entry("P6-stale-reuse".into()).or_default() += 1; + } else if !differ.is_empty() { + self.report("P6", &differ, &inc, Some(&clean), Value::Null); + } + // More oracles: object code in the rlibs, the binary, and the diagnostics. + for (kind, detail) in &others { + self.report(kind, &detail[..detail.len().min(10)], &inc, Some(&clean), Value::Null); + } + let inc_exe = inc.exe.as_ref().map(|e| e.replace(&*target.to_string_lossy(), &inc_target.to_string_lossy())); + let ra = run_exe(inc_exe.as_deref()); + let rb = run_exe(clean.exe.as_deref()); + if ra != rb { + self.report("run", &[], &inc, Some(&clean), json!({"inc": ra, "clean": rb})); + } + } + rmtree(&target); + let _ = std::fs::rename(&inc_target, &target); + self.previous = inc.rmetas; + } +} + +static SPLIT_DEBUGINFO: LazyLock = LazyLock::new(|| Regex::new(r"-Csplit-debuginfo=(packed|unpacked)").unwrap()); + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let fixture = std::fs::canonicalize(&args.fixture)?; + for f in ["STOP", "PAUSED"] { + let _ = std::fs::remove_file(work.join(f)); + } + let mut skip = vec!["rmeta"]; + if SPLIT_DEBUGINFO.is_match(&args.rustflags) { + skip.extend(["rlib", "exe"]); + } + let ctx = Ctx { + bin: fixture.file_name().unwrap().to_string_lossy().into_owned(), + args: args.clone(), + work: work.clone(), + fixture, + skip, + }; + let totals: Vec<(u64, u64, u64)> = std::thread::scope(|s| { + let handles: Vec<_> = (0..args.workers) + .map(|k| { + let ctx = &ctx; + s.spawn(move || { + let home = ctx.work.join(format!("w{k}")); + let mut w = Worker { + ctx, + k, + src: home.join("src"), + target: home.join("target"), + inc_target: home.join("target-inc"), + findings: ctx.work.join("findings"), + home, + stats: Stats::default(), + kept: BTreeMap::new(), + history: Vec::new(), + previous: BTreeMap::new(), + }; + w.run(); + (w.stats.edits, w.stats.built, w.stats.compared) + }) + }) + .collect(); + handles.into_iter().map(|h| h.join().unwrap_or_default()).collect() + }); + let sum = |f: fn(&(u64, u64, u64)) -> u64| totals.iter().map(f).sum::(); + println!("{{\"edits\": {}, \"built\": {}, \"compared\": {}}}", sum(|t| t.0), sum(|t| t.1), sum(|t| t.2)); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/fuzz_replay.rs b/crates/mirth-lab/src/tools/fuzz_replay.rs new file mode 100644 index 0000000..6b372b3 --- /dev/null +++ b/crates/mirth-lab/src/tools/fuzz_replay.rs @@ -0,0 +1,118 @@ +//! Replay a fuzz finding: apply its edits to the pristine fixture in order, with an incremental +//! build after each, then compare with a clean build. +//! +//! Prints which .rmeta files differ, and keeps the compiler's output of the last incremental +//! build and the clean build as inc.log and clean.log. --upto replays only the first N edits +//! that were kept, to find where the difference appears. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; + +use serde::Deserialize; + +use mirth_lab::cargo::{self, copy_tree, messages, relative}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + fixture: PathBuf, + #[arg(long)] + finding: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long)] + upto: Option, + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "-Zincremental-verify-ich", allow_hyphen_values = true)] + rustflags: String, + #[arg(long)] + quiet: bool, +} + +#[derive(Deserialize)] +struct Step { + edit: String, + file: String, + #[serde(default)] + before: String, + #[serde(default)] + after: String, + kept: bool, +} + +/// Whether it built, the .rmeta files by path, and the compiler's output. +fn build(args: &Args, src: &Path, target: &Path) -> (bool, BTreeMap>, String) { + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{}", args.toolchain)) + .args(["build", "--workspace", "--offline", "-j", "4", "--target-dir"]) + .arg(target) + .arg("--message-format=json-render-diagnostics") + .current_dir(src) + .env("RUSTC", &args.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + .env("RUSTFLAGS", &args.rustflags); + let Ok(r) = cargo::run_group(cmd, None) else { return (false, BTreeMap::new(), "cargo did not start".into()) }; + let rmetas = messages(&r.stdout) + .filter(|m| m.reason == "compiler-artifact") + .flat_map(|m| m.filenames) + .filter(|f| f.ends_with(".rmeta")) + .map(|f| (relative(&f, target), std::fs::read(&f).unwrap_or_default())) + .collect(); + (r.ok, rmetas, r.stderr) +} + +pub fn run(args: Args) -> anyhow::Result { + let _ = std::fs::remove_dir_all(&args.work); + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let (src, target, inc_target) = (work.join("src"), work.join("target"), work.join("target-inc")); + copy_tree(&std::fs::canonicalize(&args.fixture)?, &src, &["target", "edits", "edit"], false)?; + build(&args, &src, &target); + + let history: Vec = serde_json::from_str(&std::fs::read_to_string(args.finding.join("history.json"))?)?; + let mut kept = 0; + let mut last: Option<&Step> = None; + for step in &history { + if step.edit == "revert" { + if let Some(l) = last { + std::fs::write(src.join(&l.file), &l.before)?; + } + } else { + if args.upto.is_some_and(|u| step.kept && kept >= u) { + break; + } + std::fs::write(src.join(&step.file), &step.after)?; + last = Some(step); + kept += step.kept as usize; + } + let (ok, _, _) = build(&args, &src, &target); + if !args.quiet { + println!("{:18} {:24} {}", step.edit, step.file, if ok { "built" } else { "failed" }); + } + } + + let (ok, inc, log) = build(&args, &src, &target); + std::fs::write(work.join("inc.log"), log)?; + std::fs::rename(&target, &inc_target)?; + let (ok2, clean, log) = build(&args, &src, &target); + std::fs::write(work.join("clean.log"), log)?; + let keys: BTreeSet<&String> = inc.keys().chain(clean.keys()).collect(); + let differ: Vec<&String> = keys.into_iter().filter(|r| inc.get(*r) != clean.get(*r)).collect(); + let name = |d: &str| Path::new(d).file_name().map(|s| s.to_string_lossy().into_owned()).unwrap_or_default(); + let names: Vec = differ.iter().map(|d| name(d)).collect(); + println!( + "{{\"kept_edits\": {kept}, \"inc_ok\": {ok}, \"clean_ok\": {ok2}, \"differ\": {}}}", + serde_json::to_string(&names)?.replace("\",\"", "\", \"") + ); + for d in differ { + let n = name(d); + std::fs::write(work.join(format!("{n}.inc")), inc.get(d).map_or(&[][..], |v| v))?; + std::fs::write(work.join(format!("{n}.clean")), clean.get(d).map_or(&[][..], |v| v))?; + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/replay.rs b/crates/mirth-lab/src/tools/replay.rs new file mode 100644 index 0000000..937a8cb --- /dev/null +++ b/crates/mirth-lab/src/tools/replay.rs @@ -0,0 +1,314 @@ +//! Replay a crate's git history through incremental compilation. +//! +//! For each first-parent commit, oldest first, the workspace is checked out and built +//! incrementally on top of the previous commit's build, then built again from scratch, and the +//! two are compared: +//! +//! P6 every .rmeta Cargo reports for a workspace member is identical +//! rlib every rlib's members are identical, object code included +//! diag both builds printed the same diagnostics +//! reuse the compiler's own check of what it reused (RUSTC_VERIFY_REUSE, +//! docs/hunt/verify-reuse.patch) found nothing stale +//! ICE neither build crashed the compiler +//! split both builds succeed or both fail +//! +//! Registry dependencies are not compiled incrementally by Cargo, so the clean build starts +//! from a copy of the incremental target directory with the workspace members and the +//! incremental cache removed, and only the members are built again. Both builds use the same +//! target directory path, since Cargo derives a crate's identity from paths. +//! +//! Writes /results.jsonl, one line per commit, and keeps the logs of every problem and +//! both .rmeta files of the first few differences per crate in /findings. + +use std::collections::{BTreeMap, BTreeSet}; +use std::io::Write; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::time::Instant; + +use serde_json::json; + +use mirth_lab::cargo::{self, Collected, copy_tree, messages, relative, tail}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + repo: String, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 2000)] + commits: usize, + /// For cargo. + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "4")] + jobs: String, + /// Differences kept per crate. + #[arg(long, default_value_t = 3)] + keep: usize, + /// First commit index to replay. + #[arg(long = "from", default_value_t = 0)] + start: usize, + /// Last commit index to replay. + #[arg(long = "to")] + end: Option, + /// More flags for every build, such as -Copt-level=2. + #[arg(long, default_value = "", allow_hyphen_values = true)] + rustflags: String, + /// Do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch). + #[arg(long)] + no_verify_reuse: bool, +} + +struct Ctx<'a> { + args: &'a Args, + src: PathBuf, +} + +struct Build { + ok: bool, + ice: bool, + secs: f64, + log: String, + reuse: Vec, + untracked: BTreeSet, + rmetas: BTreeMap>, + fresh: Vec, + art: Collected, +} + +impl Ctx<'_> { + fn cmd(&self, program: &str) -> Command { + let mut c = Command::new(program); + c.current_dir(&self.src) + .env("RUSTC", &self.args.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + // A commit that denies warnings would otherwise stop building with a newer compiler. + .env("RUSTFLAGS", format!("--cap-lints=warn {}", self.args.rustflags).trim()) + .env("CARGO_TERM_COLOR", "never"); + if !self.args.no_verify_reuse { + c.env("RUSTC_VERIFY_REUSE", "1").env("RUSTC_REPORT_UNTRACKED", "1"); + } + if program == "cargo" { + c.arg(format!("+{}", self.args.toolchain)); + } + c + } + + fn run(&self, program: &str, args: &[&str]) -> (bool, String) { + match self.cmd(program).args(args).output() { + Ok(o) => (o.status.success(), String::from_utf8_lossy(&o.stdout).into_owned()), + Err(_) => (false, String::new()), + } + } + + /// Every package built from a path: workspace members and path dependencies. + fn members(&self) -> Vec { + let (mut ok, mut out) = self.run("cargo", &["metadata", "--format-version", "1"]); + if !ok { + (ok, out) = self.run("cargo", &["metadata", "--no-deps", "--format-version", "1"]); + if !ok { + return Vec::new(); + } + } + let meta: serde_json::Value = serde_json::from_str(&out).unwrap_or_default(); + let names: BTreeSet = meta["packages"] + .as_array() + .into_iter() + .flatten() + .filter(|p| p.get("source").is_none_or(|s| s.is_null())) + .filter_map(|p| p["name"].as_str().map(str::to_owned)) + .collect(); + names.into_iter().collect() + } + + /// Build, and collect the .rmeta files Cargo reports for workspace members. + fn build(&self, target: &Path) -> Build { + let t = Instant::now(); + let mut c = self.cmd("cargo"); + c.args(["build", "--lib", "-j", &self.args.jobs, "--target-dir"]) + .arg(target) + .arg("--message-format=json-render-diagnostics"); + let r = cargo::run_group(c, None).unwrap_or_else(|e| cargo::Run { + ok: false, + stdout: String::new(), + stderr: format!("cargo did not start: {e}"), + hang: false, + }); + let (mut rmetas, mut fresh) = (BTreeMap::new(), Vec::new()); + for msg in messages(&r.stdout) { + if msg.reason != "compiler-artifact" || !msg.from_path() { + continue; + } + for f in &msg.filenames { + if f.ends_with(".rmeta") { + rmetas.insert(relative(f, target), std::fs::read(f).unwrap_or_default()); + if msg.fresh { + fresh.push(msg.target_name().to_owned()); + } + } + } + } + let log = &r.stderr; + Build { + ok: r.ok, + ice: cargo::is_ice(log), + secs: (t.elapsed().as_secs_f64() * 10.0).round() / 10.0, + log: tail(log, 6000).to_owned(), + reuse: log.lines().filter(|l| l.starts_with("rustc-verify-reuse:")).map(|l| l.chars().take(3000).collect()).collect(), + untracked: cargo::untracked_reads(log), + rmetas, + fresh, + art: cargo::collect(&r.stdout, target), + } + } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let (src, target, inc_target) = (work.join("src"), work.join("target"), work.join("target-inc")); + let findings = work.join("findings"); + std::fs::create_dir_all(&findings)?; + if !src.exists() { + let ok = Command::new("git").args(["clone", "-q", &args.repo]).arg(&src).status()?.success(); + anyhow::ensure!(ok, "git clone {} failed", args.repo); + } + let ctx = Ctx { args: &args, src: src.clone() }; + let max = format!("--max-count={}", args.commits); + let mut commits: Vec = ctx.run("git", &["rev-list", "--first-parent", "--reverse", &max, "origin/HEAD"]).1.split_whitespace().map(str::to_owned).collect(); + if commits.is_empty() { + commits = ctx.run("git", &["rev-list", "--first-parent", "--reverse", &max, "HEAD"]).1.split_whitespace().map(str::to_owned).collect(); + } + let results = work.join("results.jsonl"); + let done: BTreeSet = std::fs::read_to_string(&results) + .unwrap_or_default() + .lines() + .filter_map(|l| serde_json::from_str::(l).ok()) + .filter_map(|v| v["commit"].as_str().map(str::to_owned)) + .collect(); + let mut kept: BTreeMap = BTreeMap::new(); + let keep_dir = |i: usize, commit: &str| -> PathBuf { + let d = findings.join(format!("{i:05}-{}", &commit[..10.min(commit.len())])); + let _ = std::fs::create_dir_all(&d); + d + }; + + for (i, commit) in commits.iter().enumerate() { + if done.contains(commit) || i < args.start || args.end.is_some_and(|e| i > e) { + continue; + } + ctx.run("git", &["checkout", "-q", "--force", commit]); + ctx.run("git", &["clean", "-fdxq"]); + let date = ctx.run("git", &["log", "-1", "--format=%cs", commit]).1.trim().to_owned(); + let names = ctx.members(); + let inc = ctx.build(&target); + // Reads of untracked state, each listed once in /untracked.txt. + cargo::note_untracked(&work.join("untracked.txt"), &inc.untracked); + + // The clean build: the same target directory path, starting from a copy with the + // workspace members and the incremental cache removed. + let _ = std::fs::remove_dir_all(&inc_target); + if target.exists() { + std::fs::rename(&target, &inc_target)?; + copy_tree(&inc_target, &target, &[], true)?; + } else { + std::fs::create_dir_all(&inc_target)?; + } + // Cargo refuses to clean a directory it did not create; this one may have been created + // here when the first build failed early. + std::fs::create_dir_all(&target)?; + let tag = target.join("CACHEDIR.TAG"); + if !tag.exists() { + std::fs::write(&tag, "Signature: 8a477f597d28d172789f06886806bc55\n")?; + } + let target_str = target.to_string_lossy().into_owned(); + for name in &names { + ctx.run("cargo", &["clean", "-p", name, "--target-dir", &target_str]); + } + for e in std::fs::read_dir(&target)?.flatten() { + let inc_dir = e.path().join("incremental"); + if inc_dir.is_dir() { + std::fs::remove_dir_all(inc_dir)?; + } + } + let clean = ctx.build(&target); + + let (mut problems, mut differ): (Vec, Vec) = (Vec::new(), Vec::new()); + if inc.ice || clean.ice { + problems.push("ICE".into()); + } + if inc.ok != clean.ok { + problems.push("split".into()); + } + if !inc.reuse.is_empty() { + // The compiler's own check found something it reused stale. + problems.push("reuse".into()); + std::fs::write(keep_dir(i, commit).join("reuse.txt"), inc.reuse.join("\n"))?; + } + if !clean.fresh.is_empty() { + // A member the clean build did not compile again would be compared with itself. + problems.push("stale".into()); + } + if inc.ok && clean.ok && clean.fresh.is_empty() { + let (a, b) = (&inc.rmetas, &clean.rmetas); + let keys: BTreeSet<&String> = a.keys().chain(b.keys()).collect(); + differ = keys.into_iter().filter(|r| a.get(*r) != b.get(*r)).cloned().collect(); + // More oracles: object code in the rlibs and the diagnostics. + for (kind, detail) in cargo::compare(&inc.art, &clean.art) { + if kind == "rlib" || kind == "diag" { + std::fs::write(keep_dir(i, commit).join(format!("{kind}.txt")), detail.join("\n"))?; + problems.push(kind); + } + } + if !differ.is_empty() { + problems.push("P6".into()); + for rel in &differ { + let file = Path::new(rel).file_name().map(|s| s.to_string_lossy().into_owned()).unwrap_or_default(); + let krate = file.split('-').next().unwrap_or("").to_owned(); + let n = kept.entry(krate).or_default(); + if *n < args.keep { + *n += 1; + let d = keep_dir(i, commit); + std::fs::write(d.join(format!("{file}.inc")), a.get(rel).map_or(&[][..], |v| v))?; + std::fs::write(d.join(format!("{file}.clean")), b.get(rel).map_or(&[][..], |v| v))?; + } + } + } + } + if !problems.is_empty() { + let d = keep_dir(i, commit); + std::fs::write(d.join("inc.log"), &inc.log)?; + std::fs::write(d.join("clean.log"), &clean.log)?; + } + + // Continue incrementally from the incremental build. + let _ = std::fs::remove_dir_all(&target); + std::fs::rename(&inc_target, &target)?; + let record = json!({ + "i": i, "commit": commit, "date": date, "members": names, + "inc": {"ok": inc.ok, "ice": inc.ice, "secs": inc.secs}, + "clean": {"ok": clean.ok, "ice": clean.ice, "secs": clean.secs}, + "compared": inc.rmetas.len(), "differ": differ, "problems": problems, + }); + let mut f = std::fs::OpenOptions::new().create(true).append(true).open(&results)?; + writeln!(f, "{record}")?; + let status = if inc.ok && clean.ok { + "both ok".to_owned() + } else { + format!("inc {}, clean {}", if inc.ok { "ok" } else { "failed" }, if clean.ok { "ok" } else { "failed" }) + }; + println!( + "{i:5} {date} {} {status}, {} compared, {:.1}s/{:.1}s {}", + &commit[..10.min(commit.len())], + inc.rmetas.len(), + inc.secs, + clean.secs, + problems.join(" ") + ); + } + Ok(ExitCode::SUCCESS) +} From 8a94b82763313404baf28d71ea2b81a931a376f8 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:03:59 +0000 Subject: [PATCH 19/23] coverage, coverage-compact, coverage-generators, grammar-coverage and ui-coverage scripts removed (ported to mirth-lab); docs name the subcommands Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- docs/coverage-handoff.md | 4 +- docs/coverage.md | 12 +- docs/grammar.md | 4 +- docs/hunt/guard-patterns-ignored.md | 2 +- fixtures/sink/core/src/grammar.rs | 2 +- fixtures/sink/nightly/src/lib.rs | 2 +- rustc/coverage-compact.py | 71 -------- rustc/coverage-flags.py | 2 +- rustc/coverage-generators.py | 264 ---------------------------- rustc/coverage.py | 75 -------- rustc/grammar-coverage.py | 123 ------------- rustc/ui-coverage.py | 168 ------------------ 12 files changed, 14 insertions(+), 715 deletions(-) delete mode 100644 rustc/coverage-compact.py delete mode 100644 rustc/coverage-generators.py delete mode 100644 rustc/coverage.py delete mode 100644 rustc/grammar-coverage.py delete mode 100644 rustc/ui-coverage.py diff --git a/docs/coverage-handoff.md b/docs/coverage-handoff.md index 3135b31..1e52039 100644 --- a/docs/coverage-handoff.md +++ b/docs/coverage-handoff.md @@ -68,12 +68,12 @@ $WORK` runs the ui-fulldeps tests compiletest skips at stage 1. rustc/coverage-run.sh ui-fuzz target/release/mirth-lab ui-fuzz --rustc $COV_RUSTC \ --tests $MIRTH_RUST/tests/ui --list $WORK/ui-cov/all-runnable.json --work $WORK/ui-fuzz-cov \ --edits 3 --jobs 6 # about 45 minutes -rustc/coverage-run.sh generators python3 rustc/coverage-generators.py --rustc $COV_RUSTC \ +rustc/coverage-run.sh generators target/release/mirth-lab coverage-generators --rustc $COV_RUSTC \ --rust $MIRTH_RUST --list $WORK/ui-cov/picked.json --work $WORK/gen-work --jobs 6 \ --only prints,targets,dumps,links # prints and targets: minutes; dumps: 30 minutes ``` -A new generator is a function in `coverage-generators.py` named in `--only`. A run's directory +A new generator is a function in `crates/mirth-lab/src/tools/coverage_generators.rs` named in `--only`. A run's directory under `cov-suites` is picked up by the report automatically; name it with `fulldeps` or `compiler-unit` if the programs are outside the compiler (their entry points become roots). diff --git a/docs/coverage.md b/docs/coverage.md index f3622da..791a31c 100644 --- a/docs/coverage.md +++ b/docs/coverage.md @@ -4,12 +4,12 @@ the compiler compiling it reaches. mirth-watch's `[coverage]` mode (`rustc/coverage.toml`) instruments every function and closure of the compiler's own crates (`rustc_*`) with one call at entry, which records the function's first call in each process; at exit each rustc -process writes the functions it entered. `rustc/coverage.py` joins those with the site +process writes the functions it entered. `mirth-lab coverage` joins those with the site tables. MIRTH_RUST= BUILD_DIR= MIRTH_WATCH=rustc/coverage.toml rustc/build.sh MIRTH_OUT= RUSTC=//stage1/bin/rustc cargo build # any number of builds - rustc/coverage.py --sites /mirth-sites --logs [--files] [--unhit rustc_borrowck] + target/release/mirth-lab coverage --sites /mirth-sites --logs [--files] [--unhit rustc_borrowck] The instrumented compiler is the same source as `rustc-verify12` (pinned nightly plus the local patches); 72,774 functions in 80 crates. @@ -63,7 +63,7 @@ handling), `rustc_thread_pool` (+226, `-Zthreads`). ## rustc's UI tests -`rustc/ui-coverage.py run` compiles each UI test file (the way its `//@` headers say) with the +`mirth-lab ui-coverage run` compiles each UI test file (the way its `//@` headers say) with the instrumented compiler and keeps the functions it reaches beyond sink's; `pick` chooses tests greedily. Of 20,419 files, 1,839 were skipped (auxiliary crates, other targets); together the rest reach **17,668 functions sink does not (61% of the compiler with sink's)**; **300 picked tests @@ -156,7 +156,7 @@ a normal error path and is not counted. Coverage is reported with and without th `rustc/coverage-suites.sh` runs a suite through compiletest (`./x.py test`) on the instrumented compiler, with `MIRTH_OUT` set, folding logs as they finish -(`rustc/coverage-compact.py`): compiletest handles what a standalone runner cannot (auxiliary +(`mirth-lab coverage-compact`): compiletest handles what a standalone runner cannot (auxiliary crates, every revision, `minicore` cross-targets, run-make). With `--keep-stage 0 --keep-stage 1`, and a refusal when the log shows the compiler compiling: a changed mirth runtime once made `x.py` rebuild the compiler without the instrumentation, so `build.sh` now keeps a runtime per @@ -202,8 +202,8 @@ logs the same way. | run | functions reached | |---|---:| | ui tests through incremental rebuilds (`mirth-lab ui-fuzz`, 18,553 tests, 3 edits each) | 44,196 | -| `coverage-generators.py`: every `--print` request on the host and on all 334 targets; minicore and an ABI file compiled for every target at `-Copt-level=0` and 3; the 300 picked UI tests under 48 debugging and printing options | 41,768 | -| `coverage-generators.py --only links`: a binary, cdylib, staticlib and dylib on minicore for every target with `-Clinker=true`, under 11 sets of linker options | 18,228 | +| `mirth-lab coverage-generators`: every `--print` request on the host and on all 334 targets; minicore and an ABI file compiled for every target at `-Copt-level=0` and 3; the 300 picked UI tests under 48 debugging and printing options | 41,768 | +| `mirth-lab coverage-generators --only links`: a binary, cdylib, staticlib and dylib on minicore for every target with `-Clinker=true`, under 11 sets of linker options | 18,228 | **All together, with sink and its option configurations: 50,762 of the 62,516 functions that can run (81.2%); without the 911 that only panic, 50,715 of 61,605 (82.3%).** diff --git a/docs/grammar.md b/docs/grammar.md index 6596162..6b8411a 100644 --- a/docs/grammar.md +++ b/docs/grammar.md @@ -1,11 +1,11 @@ # Covering the grammar `fixtures/sink` is the code the fuzzer, the flag walks and the history of findings run on, so -what it does not contain is never tested. `rustc/grammar-coverage.py` measures it against +what it does not contain is never tested. `mirth-lab grammar-coverage` measures it against [Ur](https://github.com/PowderworksCode/codebase/tree/main/projects/ur)'s Rust grammar (`ecosystems/rust/language/*.rsc`): - rustc/grammar-coverage.py --ur --grammar /ecosystems/rust/language fixtures/sink + target/release/mirth-lab grammar-coverage --ur --grammar /ecosystems/rust/language fixtures/sink It parses every `.rs` file with `ur parse --tree` and counts two things: diff --git a/docs/hunt/guard-patterns-ignored.md b/docs/hunt/guard-patterns-ignored.md index 841bd72..490b1aa 100644 --- a/docs/hunt/guard-patterns-ignored.md +++ b/docs/hunt/guard-patterns-ignored.md @@ -29,4 +29,4 @@ the same name (`Some(x if x > 3) | Some(x if x == 0) => x`) gives the same error **Versions.** nightly-2026-07-18 and nightly-2026-10-06 (and the local compiler). **How mirth found it.** Writing nightly syntax into `fixtures/sink` to cover every -alternative of Ur's Rust grammar (`rustc/grammar-coverage.py`); the fixture's runtime check. +alternative of Ur's Rust grammar (`mirth-lab grammar-coverage`); the fixture's runtime check. diff --git a/fixtures/sink/core/src/grammar.rs b/fixtures/sink/core/src/grammar.rs index d15ed14..269a27b 100644 --- a/fixtures/sink/core/src/grammar.rs +++ b/fixtures/sink/core/src/grammar.rs @@ -1,5 +1,5 @@ //! Stable syntax the rest of the sink does not use, found by measuring it against Ur's Rust -//! grammar (`rustc/grammar-coverage.py`): one function per construct or small group. +//! grammar (`mirth-lab grammar-coverage`): one function per construct or small group. use core::cmp::Ordering; use core::fmt::Debug; diff --git a/fixtures/sink/nightly/src/lib.rs b/fixtures/sink/nightly/src/lib.rs index 220145f..da2ff72 100644 --- a/fixtures/sink/nightly/src/lib.rs +++ b/fixtures/sink/nightly/src/lib.rs @@ -1,5 +1,5 @@ //! Nightly-only syntax: the parts of Ur's Rust grammar no stable construct reaches -//! (`rustc/grammar-coverage.py`). One module per feature, each with a function `main` checks. +//! (`mirth-lab grammar-coverage`). One module per feature, each with a function `main` checks. #![feature( auto_traits, builtin_syntax, diff --git a/rustc/coverage-compact.py b/rustc/coverage-compact.py deleted file mode 100644 index 3d808fb..0000000 --- a/rustc/coverage-compact.py +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env python3 -"""Fold coverage logs (MIRTH_OUT, from a compiler built with rustc/coverage.toml) into a running -union as they are finished, and delete them: a test suite run starts tens of thousands of rustc -processes, whose logs together would not fit on the disk. - - rustc/coverage-compact.py --logs --out [--until ] - -A log is finished when its last line is the `X` line written at exit, or when it has not changed -for ten minutes (a process that crashed). For each, /added.jsonl gets the process's source -file argument and the sites it reached that no earlier process did; /union.txt holds every -site reached so far, rewritten every pass. Runs until exists, then does a last pass. -""" - -import argparse -import json -import os -import re -import time -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--logs", required=True) -p.add_argument("--out", required=True) -p.add_argument("--until", default="") -args = p.parse_args() - -LOGS, OUT = Path(args.logs), Path(args.out) -OUT.mkdir(parents=True, exist_ok=True) -union_path = OUT / "union.txt" -union = set(union_path.read_text().split()) if union_path.exists() else set() - - -def source_of(header): - for field in header.split("\t")[3:]: - if field.endswith(".rs"): - return field - m = re.search(r"--crate-name\t(\S+)", header) - return m.group(1) if m else "" - - -def one_pass(final): - done = 0 - with (OUT / "added.jsonl").open("a") as added: - for log in list(LOGS.glob("*.log")): - try: - text = log.read_text(errors="replace") - age = time.time() - log.stat().st_mtime - except FileNotFoundError: - continue - lines = text.splitlines() - if not lines or not (lines[-1].startswith("X\t") or age > 600 or (final and age > 5)): - continue - sites = {l[2:] for l in lines if l.startswith("V\t")} - new = sites - union - union.update(new) - if new: - added.write(json.dumps({"source": source_of(lines[0]), "new": sorted(new)}) + "\n") - log.unlink() - done += 1 - union_path.write_text("\n".join(sorted(union)) + "\n") - return done - - -while True: - finishing = bool(args.until) and os.path.exists(args.until) - n = one_pass(finishing) - print(f"{time.strftime('%H:%M:%S')} {n} logs folded, {len(union)} sites", flush=True) - if finishing: - one_pass(True) - break - time.sleep(30) diff --git a/rustc/coverage-flags.py b/rustc/coverage-flags.py index 4787555..61cb958 100644 --- a/rustc/coverage-flags.py +++ b/rustc/coverage-flags.py @@ -2,7 +2,7 @@ """Coverage of the compiler across option configurations: build a fixture with a coverage-instrumented rustc (rustc/coverage.toml) once per row of a PICT transitions table, clean with the A options, then rebuilt after one random edit with the B options, each row's -rustc processes logging to /row/. Read the result with rustc/coverage.py. +rustc processes logging to /row/. Read the result with `mirth-lab coverage`. rustc/coverage-flags.py --rustc --fixture fixtures/sink --flags --table rows.tsv --out [--rows 0:40] [--workers 6] diff --git a/rustc/coverage-generators.py b/rustc/coverage-generators.py deleted file mode 100644 index c389026..0000000 --- a/rustc/coverage-generators.py +++ /dev/null @@ -1,264 +0,0 @@ -#!/usr/bin/env python3 -"""Compiler runs that rustc's test suites hardly make, for coverage (rustc/callgraph.py --gaps -lists what is left): run them with the coverage-instrumented compiler and MIRTH_OUT set, and fold -the logs with rustc/coverage-compact.py. - - MIRTH_OUT= rustc/coverage-generators.py --rustc --rust - --list picked.json --work [--jobs 8] [--only prints,targets,dumps] - -- prints: every `--print` request, on the host and on every target; -- targets: `tests/auxiliary/minicore.rs` and a file of functions with every kind of argument - and return value, compiled to an object for every target (each target's ABI, layout and - codegen code); -- links: for every target, a `no_main` binary, a cdylib, a staticlib and a dylib on minicore, - linked with `-Clinker=true` (the linker command each target's linker flavor builds, without - the linker), with linker options; -- dumps: each test of the list (rustc/ui-coverage.py pick) with each debugging and printing - option (`-Zunpretty=`, `-Zdump-mir`, `-Zprint-type-sizes`, statistics, profiling, ...). -""" - -import argparse -import json -import os -import re -import subprocess -import tempfile -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--rust", required=True) -p.add_argument("--list", required=True) -p.add_argument("--work", required=True) -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--only", default="prints,targets,dumps,links") -args = p.parse_args() -WORK = Path(args.work) -WORK.mkdir(parents=True, exist_ok=True) -ENV = dict(os.environ, RUSTC_BOOTSTRAP="1") - -PRINTS = ["all-target-specs-json", "backend-has-mnemonic", "backend-has-zstd", "calling-conventions", - "cfg", "check-cfg", "code-models", "crate-name", "crate-root-lint-levels", "deployment-target", - "file-names", "host-tuple", "link-args", "native-static-libs", "relocation-models", - "split-debuginfo", "stack-protector-strategies", "supported-crate-types", "sysroot", - "target-cpus", "target-features", "target-libdir", "target-list", "target-spec-json", - "target-spec-json-schema", "tls-models", "wasm-proc-macro-tuple"] -PER_TARGET = ["cfg", "target-spec-json", "target-cpus", "target-features", "calling-conventions", - "code-models", "relocation-models", "tls-models", "stack-protector-strategies", - "split-debuginfo", "supported-crate-types", "deployment-target", "check-cfg"] - -# Every kind of argument and return value, for each target's calling convention. -ABI = r""" -#![feature(no_core, lang_items, rustc_attrs, c_variadic, f16, f128)] -#![no_core] -#![crate_type = "lib"] -#![allow(improper_ctypes_definitions, unused)] -extern crate minicore; -use minicore::*; - -#[repr(C)] pub struct Small { a: u8, b: u16 } -#[repr(C)] pub struct Pair { a: u64, b: u64 } -#[repr(C)] pub struct Big { a: [u64; 8] } -#[repr(C)] pub struct Floats { a: f32, b: f64 } -#[repr(C)] pub struct Mixed { a: f32, b: u32 } -#[repr(C)] pub union U { a: u32, b: f32 } -#[repr(C)] pub struct Hfa { a: f32, b: f32, c: f32, d: f32 } -#[repr(C)] pub struct Empty {} -#[repr(transparent)] pub struct T(u64); -#[repr(C, packed)] pub struct Packed { a: u8, b: u32 } -#[repr(C, align(16))] pub struct Aligned { a: u8 } - -#[no_mangle] pub extern "C" fn c_small(x: Small) -> Small { x } -#[no_mangle] pub extern "C" fn c_pair(x: Pair) -> Pair { x } -#[no_mangle] pub extern "C" fn c_big(x: Big) -> Big { x } -#[no_mangle] pub extern "C" fn c_floats(x: Floats, y: f32, z: f64) -> Floats { x } -#[no_mangle] pub extern "C" fn c_mixed(x: Mixed) -> Mixed { x } -#[no_mangle] pub extern "C" fn c_union(x: U) -> U { x } -#[no_mangle] pub extern "C" fn c_hfa(x: Hfa) -> Hfa { x } -#[no_mangle] pub extern "C" fn c_empty(x: Empty) -> Empty { x } -#[no_mangle] pub extern "C" fn c_transparent(x: T) -> T { x } -#[no_mangle] pub extern "C" fn c_packed(x: Packed) -> Packed { x } -#[no_mangle] pub extern "C" fn c_aligned(x: Aligned) -> Aligned { x } -#[no_mangle] pub extern "C" fn c_ints(a: i8, b: u16, c: i32, d: u64, e: i128, f: u128, g: bool, h: char) -> i128 { e } -#[no_mangle] pub extern "C" fn c_ptrs(a: *const u8, b: &u32, c: &mut [u8; 3], f: extern "C" fn()) -> *const u8 { a } -#[no_mangle] pub extern "C" fn c_many(a: u64, b: u64, c: u64, d: u64, e: u64, f: u64, g: u64, h: u64, i: u64, j: Pair, k: f64, l: f64, m: f64, n: f64, o: f64, p: f64, q: f64, r: f64, s: f64) -> u64 { a } -#[no_mangle] pub unsafe extern "C" fn c_variadic(a: u32, mut args: ...) -> u32 { a } -pub fn rust_all(a: Small, b: Pair, c: Big, d: Floats, e: (u8, u64), f: [u32; 5], g: &[u8], h: &str, i: u128) -> Big { c } -pub fn rust_f16(a: f16, b: f128) -> f128 { b } -#[no_mangle] pub extern "C" fn c_f16(a: f16, b: f128) -> f128 { b } -#[no_mangle] pub extern "system" fn system(a: Pair) -> Pair { a } -#[no_mangle] pub extern "C-unwind" fn c_unwind(a: Pair) -> Pair { a } -pub static TABLE: [extern "C" fn(Pair) -> Pair; 2] = [c_pair, c_unwind_shim]; -extern "C" fn c_unwind_shim(a: Pair) -> Pair { a } -extern "C" { fn imported(a: Big, b: Floats) -> Hfa; } -pub unsafe fn call_imported(a: Big, b: Floats) -> Hfa { imported(a, b) } -""" - - -def run(argv, cwd=None, timeout=300): - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=timeout, cwd=cwd, env=ENV) - return r.returncode, r.stderr - except subprocess.TimeoutExpired: - return -1, "timeout" - - -def targets(): - out = subprocess.run([args.rustc, "--print", "target-list"], capture_output=True, text=True, env=ENV) - return out.stdout.split() - - -def prints(): - jobs = [[args.rustc, "--print", kind, "-Zunstable-options", "-"] for kind in PRINTS] - for target in targets(): - for kind in PER_TARGET: - jobs.append([args.rustc, "--print", kind, "--target", target, "-Zunstable-options", "-"]) - with tempfile.TemporaryDirectory(dir=WORK) as d: - empty = Path(d) / "lib.rs" - empty.write_text("") - jobs = [[a if a != "-" else str(empty) for a in j] for j in jobs] - with ThreadPoolExecutor(args.jobs) as ex: - done = list(ex.map(lambda j: run(j, cwd=d), jobs)) - print(f"prints: {len(jobs)} runs, {sum(c == 0 for c, _ in done)} succeeded", flush=True) - - -def one_target(target): - with tempfile.TemporaryDirectory(dir=WORK) as d: - minicore = Path(args.rust) / "tests/auxiliary/minicore.rs" - (Path(d) / "abi.rs").write_text(ABI) - results = [] - base = [args.rustc, "--target", target, "-Zunstable-options", "--edition", "2021", - "-Cpanic=abort", "--out-dir", d] - code, err = run(base + ["--crate-type", "rlib", "--crate-name", "minicore", "-Copt-level=1", - "--emit=link,obj", str(minicore)], cwd=d) - results.append(code) - if code == 0: - for opt in ("0", "3"): - code, err = run(base + ["--emit=obj,asm,llvm-ir", f"-Copt-level={opt}", "-Cdebuginfo=2", - "--extern", f"minicore={d}/libminicore.rlib", "abi.rs"], cwd=d) - results.append(code) - return target, results - - -def cross(): - with ThreadPoolExecutor(args.jobs) as ex: - done = list(ex.map(one_target, targets())) - built = sum(all(c == 0 for c in r) and len(r) == 3 for _, r in done) - print(f"targets: {len(done)} targets, {built} built minicore and the ABI file", flush=True) - (WORK / "targets.json").write_text(json.dumps(dict(done), indent=0)) - - -LINKED = r""" -#![feature(no_core, lang_items)] -#![no_core] -#![no_main] -extern crate minicore; -#[no_mangle] pub extern "C" fn exported(a: u32) -> u32 { a } -#[no_mangle] pub static DATA: u32 = 7; -#[link(name = "c")] extern "C" { fn puts(p: *const u8) -> i32; } -#[link(name = "m", kind = "static")] extern "C" {} -#[link(name = "framework_like", kind = "dylib", modifiers = "+verbatim")] extern "C" {} -""" -LINK_OPTIONS = [[], ["-Cprefer-dynamic", "-Crelocation-model=pic"], ["-Cstrip=symbols", "-Clink-dead-code"], - ["-Clink-self-contained=yes"], ["-Cdebuginfo=2", "-Csplit-debuginfo=packed"], - ["-Clink-arg=-Wl,--foo", "-Clink-args=-x -y", "-Zpre-link-args=-z"], - ["-Cdefault-linker-libraries", "-Zlink-native-libraries=no"], ["-Ccontrol-flow-guard"], - ["-Zstaticlib-allow-rdylib-deps"], ["-Copt-level=s", "-Clto=fat"], ["-Ccode-model=large"]] - - -def one_link(target): - with tempfile.TemporaryDirectory(dir=WORK) as d: - minicore = Path(args.rust) / "tests/auxiliary/minicore.rs" - (Path(d) / "linked.rs").write_text(LINKED) - base = [args.rustc, "--target", target, "-Zunstable-options", "--edition", "2021", "-Cpanic=abort", - "--out-dir", d, "-Clinker=true"] - code, _ = run(base + ["--crate-type", "rlib", "--crate-name", "minicore", "--emit=link", - str(minicore)], cwd=d) - ok = 0 - if code == 0: - for options in LINK_OPTIONS: - for kind in ("bin", "cdylib", "staticlib", "dylib"): - c, _ = run(base + ["--crate-type", kind, "--extern", f"minicore={d}/libminicore.rlib", - "-Csave-temps", *options, "linked.rs"], cwd=d) - ok += c == 0 - return ok - - -def links(): - with ThreadPoolExecutor(args.jobs) as ex: - done = list(ex.map(one_link, targets())) - print(f"links: {len(done)} targets, {sum(done)} links succeeded", flush=True) - - -DUMPS = [[f"-Zunpretty={m}"] for m in ("normal", "expanded", "expanded,identified", "expanded,hygiene", - "ast-tree", "ast-tree,expanded", "hir", "hir,identified", - "hir,typed", "hir-tree", "thir-tree", "thir-flat", "mir", - "stable-mir", "mir-cfg")] + [ - ["-Zdump-mir=all", "-Zdump-mir-dataflow", "-Zdump-mir-graphviz", "-Zmir-include-spans=on"], - ["-Zprint-type-sizes"], ["-Zprint-mono-items=yes", "--emit=link"], ["-Zmeta-stats"], ["-Zhir-stats"], - ["-Zinput-stats"], ["-Zself-profile", "-Zself-profile-events=all"], ["-Ztime-passes"], - ["-Zquery-dep-graph", "-Zdump-dep-graph", "-Cincremental=inc"], ["-Zincremental-info", "-Cincremental=inc"], - ["-Zdump-mono-stats", "-Zdump-mono-stats-format=json", "--emit=link"], ["-Zprint-codegen-stats", "--emit=link"], - ["-Zvalidate-mir", "-Zlint-mir", "-Zmir-opt-level=4"], ["-Zverbose-internals", "-Zidentify-regions"], - ["-Ztrack-diagnostics", "-Zteach"], ["-Zthreads=4"], ["-Zpolonius=next"], ["-Zinline-mir", "-Zmir-opt-level=3"], - ["-Zrandomize-layout"], ["-Zwrite-long-types-to-disk=no", "-Zverbose-internals"], - ["-Zunleash-the-miri-inside-of-you"], ["-Zno-analysis"], ["-Zprofile-closures"], ["-Zui-testing"], - ["-Cinstrument-coverage", "--emit=link"], ["-Zemit-stack-sizes", "--emit=link"], - ["--error-format=json", "--json=diagnostic-rendered-ansi,artifacts,future-incompat,unused-externs"], - ["--error-format=human-annotate-rs"], ["--error-format=short"], ["-Zterminal-urls=yes", "--color=always"], - ["-Wunused", "-Wrust-2018-idioms", "-Wrust-2021-compatibility", "-Wrust-2024-compatibility", "-Wclippy::all"], - ["-Fwarnings", "--cap-lints=warn"], ["-Zcodegen-source-order", "--emit=link"], -] - - -def headers(text): - flags, edition, revision = [], None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - if revision: - flags += ["--cfg", revision] - return flags, edition - - -def dump(test): - path = Path(args.rust) / "tests/ui" / test - text = path.read_text(errors="replace") - flags, edition = headers(text) - ok = 0 - for extra in DUMPS: - with tempfile.TemporaryDirectory(dir=WORK) as d: - emit = [] if any(e.startswith("--emit") for e in extra) else ["--emit=metadata"] - code, _ = run([args.rustc, str(path), "--edition", edition or "2015", *emit, "--out-dir", d, - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", - *flags, *extra], cwd=d, timeout=120) - ok += code == 0 - return ok - - -def dumps(): - picked = json.loads(Path(args.list).read_text()) - tests = [t["test"] if isinstance(t, dict) else t for t in picked] - with ThreadPoolExecutor(args.jobs) as ex: - done = list(ex.map(dump, tests)) - print(f"dumps: {len(tests)} tests x {len(DUMPS)} options, {sum(done)} runs succeeded", flush=True) - - -only = args.only.split(",") -if "prints" in only: - prints() -if "targets" in only: - cross() -if "dumps" in only: - dumps() -if "links" in only: - links() diff --git a/rustc/coverage.py b/rustc/coverage.py deleted file mode 100644 index 161a67c..0000000 --- a/rustc/coverage.py +++ /dev/null @@ -1,75 +0,0 @@ -#!/usr/bin/env python3 -"""Which of the compiler's functions ran, from a compiler built with rustc/coverage.toml. - - rustc/coverage.py --sites /mirth-sites --logs [--logs ...] - [--json out.json] [--files] [--unhit ] - -The site tables list each instrumented function (`cover` sites: id, crate, path, span); every -rustc process run with MIRTH_OUT set writes, at exit, a `V ` line for each function it -entered. This reads both and prints, per crate, how many functions ran; with --files, per -source file; with --unhit, the functions that never ran, in the crates or files matching. - -Several --logs directories (several runs: fixtures, flags) are combined. -""" - -import argparse -import json -from collections import defaultdict -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--sites", required=True) -p.add_argument("--logs", action="append", required=True) -p.add_argument("--json") -p.add_argument("--files", action="store_true") -p.add_argument("--unhit", action="append", default=[]) -args = p.parse_args() - -functions = {} # site -> (crate, path, span) -for table in Path(args.sites).glob("*.sites"): - for line in table.read_text(errors="replace").splitlines(): - f = line.split("\t") - if len(f) >= 7 and f[1] == "cover": - functions[f[0]] = (f[3], f[4], f[6]) - -hit = set() -processes = 0 -for directory in args.logs: - for log in Path(directory).glob("*.log"): - processes += 1 - for line in log.read_text(errors="replace").splitlines(): - if line.startswith("V\t"): - hit.add(line[2:]) - -unknown = hit - functions.keys() -by_crate = defaultdict(lambda: [0, 0]) -by_file = defaultdict(lambda: [0, 0]) -for site, (krate, path, span) in functions.items(): - file = span.rsplit(":", 2)[0] - ran = site in hit - for table, key in ((by_crate, krate), (by_file, file)): - table[key][0] += 1 - table[key][1] += ran - -total, ran = len(functions), len(hit & functions.keys()) -print(f"{processes} processes; {ran} of {total} functions ran ({100 * ran / max(total, 1):.1f}%)" - + (f"; {len(unknown)} sites not in the tables" if unknown else "")) -print(f"{'crate':40} {'ran':>7} {'of':>7} {'%':>6}") -for krate, (n, r) in sorted(by_crate.items(), key=lambda kv: kv[1][1] / kv[1][0]): - print(f"{krate:40} {r:7} {n:7} {100 * r / n:6.1f}") -if args.files: - print() - print(f"{'file':80} {'ran':>6} {'of':>6}") - for file, (n, r) in sorted(by_file.items(), key=lambda kv: (kv[1][1] / kv[1][0], -kv[1][0])): - print(f"{file:80} {r:6} {n:6}") -for pattern in args.unhit: - print(f"\nnever ran, matching {pattern!r}:") - for site, (krate, path, span) in sorted(functions.items(), key=lambda kv: kv[1][2]): - if site not in hit and (pattern in krate or pattern in span): - print(f" {path} {span}") -if args.json: - Path(args.json).write_text(json.dumps({ - "processes": processes, "functions": total, "ran": ran, - "crates": {k: {"functions": n, "ran": r} for k, (n, r) in by_crate.items()}, - "files": {k: {"functions": n, "ran": r} for k, (n, r) in by_file.items()}, - "unhit": sorted(f"{functions[s][1]}\t{functions[s][2]}" for s in functions.keys() - hit)}, indent=1)) diff --git a/rustc/grammar-coverage.py b/rustc/grammar-coverage.py deleted file mode 100644 index 3c85ebb..0000000 --- a/rustc/grammar-coverage.py +++ /dev/null @@ -1,123 +0,0 @@ -#!/usr/bin/env python3 -"""Which parts of the Rust grammar a fixture's sources use, by Ur's grammar. - - rustc/grammar-coverage.py --ur --grammar /ecosystems/rust/language \\ - [--json out.json] [--edition 2024] - -Ur's Rust grammar (Urscal modules, `syntax Sort = Label: ... | Label: ... | Other ;`) names -each alternative of each syntax sort. `ur parse --tree` prints a file's tree with every node as -`(Sort::Label ...)` and every literal token in quotes. Two measures: - - alternatives each labeled alternative, by construct: the label within its module, with - Conditions.rsc (expressions in condition position) counted as Expressions - literals each keyword and operator the syntax rules mention, used or not - -Prints what is missing, by module, and a summary. Lexical rules, layout and keyword lists are -not syntax alternatives and are left out. -""" - -import argparse -import json -import re -import subprocess -from collections import defaultdict -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--ur", required=True) -p.add_argument("--grammar", required=True) -p.add_argument("fixture") -p.add_argument("--json") -args = p.parse_args() - - -def strip_comments(text): - return re.sub(r"//[^\n]*", "", text) - - -STRING = r'"(?:[^"\\]|\\.)*"' - - -def syntax_rules(text): - """(sort, body) for each `syntax` rule of a module: from `syntax Name ... =` to the `;` - that ends it, outside quotes and brackets.""" - for m in re.finditer(r"(?m)^syntax\s+(\w+)[^=]*=", text): - i, depth = m.end(), 0 - while i < len(text): - c = text[i] - if c == '"': - q = re.match(STRING, text[i:]) - i += q.end() if q else 1 - continue - if c in "([{": - depth += 1 - elif c in ")]}": - depth -= 1 - elif c == ";" and depth == 0: - break - i += 1 - yield m.group(1), text[m.end():i] - - -# Conditions.rsc repeats the expression sorts for condition position (no struct literals): the -# same constructs, so counted with the expressions. -FAMILY = {"Conditions": "Expressions"} - -grammar = {} # (family, label) -> [sorts] -literals = {} # literal -> family -for f in sorted(Path(args.grammar).glob("*.rsc")): - if f.stem in ("Testing", "Semantics", "Language", "Rust"): - continue - family = FAMILY.get(f.stem, f.stem) - text = strip_comments(f.read_text()) - for sort, body in syntax_rules(text): - body = re.sub(r'@\w+=' + STRING, "", body) - for lit in re.findall(STRING, body): - lit = lit[1:-1].replace("\\<", "<").replace("\\>", ">").replace('\\"', '"').replace("\\\\", "\\") - literals.setdefault(lit, family) - bare = re.sub(STRING, '""', body) - bare = re.sub(r'@\w+="[^"]*"', "", bare) - for label in re.findall(r"(? --tests /tests/ui - --sites /mirth-sites --baseline [--baseline ...] --out [--jobs 6] - rustc/ui-coverage.py pick --out [--count 200] - -`run` compiles each test file the way its `//@` headers say, as far as one rustc call can: -`compile-flags`, `edition`, the first of `revisions` (as `--cfg` with its own flags), metadata -only for tests that do not build (check-pass, and tests expected to fail before codegen), a -full build otherwise. Tests that need auxiliary crates, proc macros, another target or -`minicore` are skipped. Writes /tests.jsonl: per test, whether it compiled and the -indices (into /functions.json) of the functions it reached beyond the baseline. - -`pick` chooses tests greedily, each adding the most functions not yet reached, and writes -/picked.json. -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -import sys -import tempfile -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -sub = p.add_subparsers(dest="cmd", required=True) -r = sub.add_parser("run") -r.add_argument("--rustc", required=True) -r.add_argument("--tests", required=True) -r.add_argument("--sites", required=True) -r.add_argument("--baseline", action="append", default=[]) -r.add_argument("--out", required=True) -r.add_argument("--jobs", type=int, default=6) -r.add_argument("--limit", type=int, default=0) -k = sub.add_parser("pick") -k.add_argument("--out", required=True) -k.add_argument("--count", type=int, default=200) -args = p.parse_args() -OUT = Path(args.out) - -SKIP = re.compile(r"^//@\s*(aux-build|aux-crate|aux-bin|proc-macro|add-minicore|needs-llvm-components|" - r"only-(?!x86_64|linux|unix|64bit)|ignore-x86_64|ignore-linux|needs-sanitizer|needs-profiler|" - r"needs-asm-support|known-bug)", re.M) -NO_BUILD = ("check-pass", "check-fail") - - -def hits(directory): - out = set() - for log in Path(directory).glob("*.log"): - for line in log.read_text(errors="replace").splitlines(): - if line.startswith("V\t"): - out.add(line[2:]) - return out - - -def headers(text): - flags, edition, revision, kind = [], None, None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - elif key in ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail"): - kind = key - if revision: - flags += ["--cfg", revision] - return flags, edition, kind - - -def run_one(path): - text = path.read_text(errors="replace") - rel = str(path.relative_to(args.tests)) - if SKIP.search(text): - return {"test": rel, "status": "skipped"} - flags, edition, kind = headers(text) - with tempfile.TemporaryDirectory(dir=OUT / "scratch") as d: - emit = "--emit=metadata" if kind in NO_BUILD or kind is None else "--emit=link" - argv = [args.rustc, str(path), "--edition", edition or "2015", emit, "--out-dir", d, - "-Zunstable-options", "-Ainternal_features", *flags] - env = dict(os.environ, MIRTH_OUT=d + "/logs", RUSTC_BOOTSTRAP="1") - try: - done = subprocess.run(argv, capture_output=True, text=True, timeout=120, env=env, cwd=d) - ice = ("internal compiler error" in done.stderr or "the compiler unexpectedly panicked" in done.stderr - or "rustc interrupted by SIG" in done.stderr) - status = "ok" if done.returncode == 0 else ("ice" if ice else "error") - first = next((l for l in done.stderr.splitlines() if l.startswith("error")), "")[:160] - except subprocess.TimeoutExpired: - status, first = "timeout", "" - new = sorted(index[s] for s in hits(d + "/logs") - baseline if s in index) - return {"test": rel, "status": status, "kind": kind, "error": first, "new": new} - - -if args.cmd == "run": - functions = [] - for table in Path(args.sites).glob("*.sites"): - for line in table.read_text(errors="replace").splitlines(): - f = line.split("\t") - if len(f) >= 7 and f[1] == "cover": - functions.append((f[0], f[4], f[6])) - functions.sort() - index = {site: i for i, (site, _, _) in enumerate(functions)} - baseline = set() - for b in args.baseline: - for sub_dir in [Path(b), *Path(b).glob("*")]: - if sub_dir.is_dir(): - baseline |= hits(sub_dir) - OUT.mkdir(parents=True, exist_ok=True) - (OUT / "scratch").mkdir(exist_ok=True) - (OUT / "functions.json").write_text(json.dumps( - {"functions": [[path, span] for _, path, span in functions], - "baseline": sorted(index[s] for s in baseline if s in index)})) - done = set() - results = OUT / "tests.jsonl" - if results.exists(): - done = {json.loads(l)["test"] for l in results.read_text().splitlines()} - tests = sorted(p for p in Path(args.tests).rglob("*.rs") - if "auxiliary" not in p.parts and str(p.relative_to(args.tests)) not in done) - if args.limit: - tests = tests[:args.limit] - print(f"{len(functions)} functions, {len(baseline & index.keys())} in the baseline; " - f"{len(tests)} tests to run", flush=True) - with ThreadPoolExecutor(args.jobs) as ex, results.open("a") as out: - for n, res in enumerate(ex.map(run_one, tests)): - out.write(json.dumps(res) + "\n") - if n % 500 == 0: - out.flush() - print(f"{n} tests", flush=True) - shutil.rmtree(OUT / "scratch", ignore_errors=True) - -if args.cmd == "pick": - info = json.loads((OUT / "functions.json").read_text()) - tests = [json.loads(l) for l in (OUT / "tests.jsonl").read_text().splitlines()] - tests = [t for t in tests if t.get("new")] - reached = set() - for t in tests: - reached |= set(t["new"]) - covered, picked = set(), [] - sets = {t["test"]: set(t["new"]) for t in tests} - status = {t["test"]: t["status"] for t in tests} - while len(picked) < args.count and sets: - best = max(sets, key=lambda name: len(sets[name] - covered)) - gain = sets[best] - covered - if not gain: - break - covered |= gain - picked.append({"test": best, "status": status[best], "adds": len(gain), "total": len(covered)}) - del sets[best] - total = len(info["functions"]) - base = len(info["baseline"]) - print(f"baseline {base} of {total} functions ({100 * base / total:.1f}%); all tests reach " - f"{len(reached)} more; {len(picked)} picked tests reach {len(covered)} more " - f"({100 * (base + len(covered)) / total:.1f}% in all)") - for t in picked[:40]: - print(f" +{t['adds']:5} {t['total']:6} {t['status']:7} {t['test']}") - (OUT / "picked.json").write_text(json.dumps(picked, indent=1)) From cde950ad593b2dbdab47791a2f054053808e5a5a Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:09:52 +0000 Subject: [PATCH 20/23] rustc/coverage-flags.py removed (ported to mirth-lab coverage-flags) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- rustc/coverage-flags.py | 98 ----------------------------------------- 1 file changed, 98 deletions(-) delete mode 100644 rustc/coverage-flags.py diff --git a/rustc/coverage-flags.py b/rustc/coverage-flags.py deleted file mode 100644 index 61cb958..0000000 --- a/rustc/coverage-flags.py +++ /dev/null @@ -1,98 +0,0 @@ -#!/usr/bin/env python3 -"""Coverage of the compiler across option configurations: build a fixture with a -coverage-instrumented rustc (rustc/coverage.toml) once per row of a PICT transitions table, -clean with the A options, then rebuilt after one random edit with the B options, each row's -rustc processes logging to /row/. Read the result with `mirth-lab coverage`. - - rustc/coverage-flags.py --rustc --fixture fixtures/sink - --flags --table rows.tsv --out [--rows 0:40] [--workers 6] -""" - -import argparse -import csv -import importlib.util -import json -import os -import random -import shutil -import subprocess -import sys -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -here = Path(__file__).parent -sys.path.insert(0, str(here)) -import mutations # noqa: E402 - -spec = importlib.util.spec_from_file_location("flag_model", here / "flag-model.py") -flag_model = importlib.util.module_from_spec(spec) -spec.loader.exec_module(flag_model) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--fixture", required=True) -p.add_argument("--flags", required=True) -p.add_argument("--table", required=True) -p.add_argument("--out", required=True) -p.add_argument("--rows", default="") -p.add_argument("--workers", type=int, default=6) -p.add_argument("--toolchain", default="nightly-2026-10-06") -args = p.parse_args() - -OUT = Path(args.out).resolve() -opts = {flag_model.pname(o["flag"] + o["name"]): o for o in json.load(open(Path(args.flags) / "options.json"))} -rows = list(csv.DictReader(open(args.table), delimiter="\t")) -idx = list(range(len(rows))) -if args.rows: - lo, hi = (int(x) if x else None for x in args.rows.split(":")) - idx = idx[lo:hi] - - -def flags(row, side): - out = [flag_model.FLAG_BASE] - for k, v in row.items(): - if k.startswith(side + "_") and v != "absent": - o = opts[k[2:]] - f = o["flag"] + o["name"] - out.append(f if v == "present" else f + "=" + v.replace(";", ",")) - return out - - -def run(i): - logs = OUT / f"row{i}" - if logs.exists(): - return i, "done before" - work = OUT / f"work{i}" - shutil.rmtree(work, ignore_errors=True) - src = work / "s" - shutil.copytree(args.fixture, src, ignore=shutil.ignore_patterns("target")) - results = [] - for side in "AB": - if side == "B": - rng = random.Random(i) - paths = sorted(p for p in src.rglob("*.rs") if "target" not in p.parts) - for _ in range(20): - path = rng.choice(paths) - fn = rng.choices([e for e, _ in mutations.EDITS], weights=[w for _, w in mutations.EDITS])[0] - if path.name == "build.rs": - continue - new = fn(path.read_text(), rng, 0) - if new is not None: - path.write_text(new) - break - e = dict(os.environ, RUSTC=args.rustc, RUSTC_WRAPPER="", CARGO_INCREMENTAL="1", - RUSTFLAGS=" ".join(flags(rows[i], side)), MIRTH_OUT=str(work / "logs")) - r = subprocess.run(["cargo", f"+{args.toolchain}", "build", "--workspace", "--offline", "-j", "4", - "--target", "x86_64-unknown-linux-gnu", "--target-dir", str(work / "t")], - cwd=src, env=e, capture_output=True, text=True, timeout=1800) - results.append("ok" if r.returncode == 0 else "failed") - (work / "logs").mkdir(exist_ok=True) - (work / "logs").rename(logs) - shutil.rmtree(work, ignore_errors=True) - return i, " ".join(results) - - -OUT.mkdir(parents=True, exist_ok=True) -with ThreadPoolExecutor(args.workers) as ex: - for i, result in ex.map(run, idx): - print(f"row {i}: {result}", flush=True) From 6a16adf4e93862caadeac577e8297cbbec7c4825 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:11:22 +0000 Subject: [PATCH 21/23] xlink: a timed-out build is its own result class Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- crates/mirth-lab/src/tools/xlink.rs | 2 ++ 1 file changed, 2 insertions(+) diff --git a/crates/mirth-lab/src/tools/xlink.rs b/crates/mirth-lab/src/tools/xlink.rs index 526c9ef..79e4529 100644 --- a/crates/mirth-lab/src/tools/xlink.rs +++ b/crates/mirth-lab/src/tools/xlink.rs @@ -126,6 +126,8 @@ fn one(args: &Args, probe: &Path, target: &str) -> Res { }; let result = if exit == Exit::Code(0) { "ok" + } else if exit == Exit::Timeout { + "timeout" } else if err.contains("internal compiler error") || err.contains("panicked at") { "ice" } else if err.contains("linking with") || err.contains("lld: error") { From e9f04a30b681c3fea7552b6aed042f1c18503a8e Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:22:47 +0000 Subject: [PATCH 22/23] Finding 32: new-solver compile-time regression on long iterator chains, bisected to nightly-2026-08-04 (#160254 the only solver PR in range); facts and the 200-map test Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- docs/hunt.md | 1 + docs/hunt/iter-chain-solver-regression.md | 69 +++++++++++++++++++++++ docs/hunt/tests/iter-chain-200.rs | 1 + 3 files changed, 71 insertions(+) create mode 100644 docs/hunt/iter-chain-solver-regression.md create mode 100644 docs/hunt/tests/iter-chain-200.rs diff --git a/docs/hunt.md b/docs/hunt.md index 7864e8f..e52fcc5 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -55,6 +55,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 29 | machine-applicable lint fixes (what `cargo fix` applies unasked) break builds: `unused_variables` turns `ref b` into a moving `_b` and renames only the declaration of variables mentioned elsewhere, `unused_mut` changes one or-pattern alternative or a variable a `move` closure assigns, `unused_imports` removes a glob that resolution needs | **looks new**; stable 1.98; found by the suggestions-apply check (111 lint fixes in UI tests); six 3–7 line reductions; [facts](hunt/lint-fixes-break-builds.md) | | 30 | compiler-internal debug output in user-facing diagnostics: under the default (new) solver, E0308 help suggests `as fn(?0t) -> ?0t`; an E0391 cycle note prints `Binder { value: ConstEvaluatable(AliasConst(… DefId(0:7 ~ …` (blessed in `offset-of/inside-array-length.stderr`) | low, diagnostics; found by the diagnostic-invariants check over 18,374 UI tests (excluding tests that ask for verbose output); the first not in CI because of the solver pin ([`solver-triage.md`](solver-triage.md) item I) | | 31 | `#[rustc_main]` on a struct, impl, trait or module, on stable: after the expected E0658 and "cannot be used on structs", rustc ICEs ("unexpected sort of node in fn_sig()", `collect.rs`): the item is still taken as the entry point | **looks new**, low (error recovery, internal attribute); regression between 1.91.0 and 1.93.0; found by the feature-gate check; [repro](hunt/tests/rustc-main-on-struct.rs) | +| 32 | new-solver compile-time regression: a chain of N `.map()` calls type-checks in 4.5 s / 520 MB at N=200 on nightly-2026-08-03 and 37–58 s / 2.0–2.9 GB from nightly-2026-08-04, with a new "overflow evaluating the requirement `Map<…>: Iterator`" future-compat warning; nightly's default solver is the new one, so default builds regressed from 2.2 s (old solver, July) to 53 s | **looks new**, medium (compile time, realistic code shape); bisected over nightlies to #160254 (the only solver PR in the range); found by the scaling check; [facts](hunt/iter-chain-solver-regression.md) | Findings 1 and 2 are single-threaded: an ordinary `cargo build`, an edit, another `cargo build`, and the metadata differs from a clean build of the edited source. Both come diff --git a/docs/hunt/iter-chain-solver-regression.md b/docs/hunt/iter-chain-solver-regression.md new file mode 100644 index 0000000..13a04d5 --- /dev/null +++ b/docs/hunt/iter-chain-solver-regression.md @@ -0,0 +1,69 @@ +# New solver: long iterator chains take 10× longer and 4× the memory since nightly-2026-08-04 + +Facts for finding 32. Found by the scaling check (`mirth-lab scale-check`, shape `iter-chain`: +growth exponent of compile time 2.6–3.5 at N ≤ 100). + +## What happens + +[`tests/iter-chain-200.rs`](tests/iter-chain-200.rs) is one statement: +`(0u64..10).map(|x| x.wrapping_add(0)) … .map(|x| x.wrapping_add(199)).sum()`, 200 `map` +calls. User time and peak memory of `rustc iter-chain-200.rs` (-Copt-level=0): + +| toolchain | default | `-Znext-solver=coherence` (old solver) | `-Znext-solver=globally` | +|---|---|---|---| +| 1.80.0 | 5.7 s, 115 MB | | | +| 1.90.0 | 7.0 s, 125 MB | | | +| 1.98.0 | 2.3 s, 118 MB | | | +| nightly-2026-07-18 | 2.2 s, 119 MB | 2.2 s, 118 MB | 4.6 s, 517 MB | +| nightly-2026-08-03 | | | 4.5 s, 519 MB | +| nightly-2026-08-04 | | | 37.3 s, 2.42 GB | +| nightly-2026-10-06 | 53.4 s, 2.04 GB | 6.1 s, 131 MB | 52.8 s, 2.04 GB | + +At the mirth pin (default solver): N=50 0.19 s, N=100 1.2 s, N=200 68.6 s and 2.0 GB. +`-Ztime-passes` puts 65 of 73 s in `type_check_crate`. From nightly-2026-08-04 the build also +warns once, "overflow evaluating the requirement `Map…>>: Iterator`" +("this was previously accepted by the compiler but is being phased out"); earlier nightlies +do not warn. + +Two separate changes show in the table: the new-solver slowdown between 2026-08-03 and +2026-08-04 (this finding), and the default switching to the new solver between July and +October ([`solver.md`](../solver.md)). The old solver also went from 2.2 s to 6.1 s over the +same period; not bisected. + +## Bisection + +Over nightlies, `-Znext-solver=globally`, "bad" above 20 s user time: 2026-07-28 4.5 s, +2026-08-02 4.7 s, 2026-08-03 4.5 s, 2026-08-04 37.3 s, 2026-08-07 38.0 s, 2026-08-27 58.4 s. + +- last good: nightly-2026-08-03, 11177f2235f0c842b00f82c558ad9480c0c3a895 +- first bad: nightly-2026-08-04, 504869653f510b279c542e65ccd1ea9710c119ba + +The range has four merges. One touches the trait solver: rollup #160414, containing #160254 +("fix-fcw-missing", commit 1489e477b62 "rerun even if the goal has ty vars"). Not confirmed by +building a compiler with it reverted. + +## Where + +`compiler/rustc_next_trait_solver/src/solve/eval_ctxt/mod.rs`, +`maybe_evaluate_root_goal_with_higher_recursion_limit` (and the proof-tree variant). #160254 +removed this early return: + +```rust + // Some goals no longer overflow after the stalled infers are resolved. + // Thus we don't have to rerun eagerly here. + let has_stalled_infers = match predicate.kind().skip_binder() { … }; + if has_stalled_infers { + return; + } +``` + +so a root goal that overflows while it still has inference variables is now re-evaluated with +twice the recursion limit (to decide whether to emit the overflow future-compat warning). The +goal in the warning has one (`Map<_, …>`, innermost). Whether the rerun happens once per +fulfillment iteration, which would fit the growth with N, was not checked. + +## Scope + +Any method chain long enough that the solver overflows on the receiver's trait goal while its +innermost type is still being inferred. 200 is long for hand-written code; generated code and +builder- or iterator-heavy macros reach it. diff --git a/docs/hunt/tests/iter-chain-200.rs b/docs/hunt/tests/iter-chain-200.rs new file mode 100644 index 0000000..075917d --- /dev/null +++ b/docs/hunt/tests/iter-chain-200.rs @@ -0,0 +1 @@ +fn main() { let s: u64 = (0u64..10).map(|x| x.wrapping_add(0)).map(|x| x.wrapping_add(1)).map(|x| x.wrapping_add(2)).map(|x| x.wrapping_add(3)).map(|x| x.wrapping_add(4)).map(|x| x.wrapping_add(5)).map(|x| x.wrapping_add(6)).map(|x| x.wrapping_add(7)).map(|x| x.wrapping_add(8)).map(|x| x.wrapping_add(9)).map(|x| x.wrapping_add(10)).map(|x| x.wrapping_add(11)).map(|x| x.wrapping_add(12)).map(|x| x.wrapping_add(13)).map(|x| x.wrapping_add(14)).map(|x| x.wrapping_add(15)).map(|x| x.wrapping_add(16)).map(|x| x.wrapping_add(17)).map(|x| x.wrapping_add(18)).map(|x| x.wrapping_add(19)).map(|x| x.wrapping_add(20)).map(|x| x.wrapping_add(21)).map(|x| x.wrapping_add(22)).map(|x| x.wrapping_add(23)).map(|x| x.wrapping_add(24)).map(|x| x.wrapping_add(25)).map(|x| x.wrapping_add(26)).map(|x| x.wrapping_add(27)).map(|x| x.wrapping_add(28)).map(|x| x.wrapping_add(29)).map(|x| x.wrapping_add(30)).map(|x| x.wrapping_add(31)).map(|x| x.wrapping_add(32)).map(|x| x.wrapping_add(33)).map(|x| x.wrapping_add(34)).map(|x| x.wrapping_add(35)).map(|x| x.wrapping_add(36)).map(|x| x.wrapping_add(37)).map(|x| x.wrapping_add(38)).map(|x| x.wrapping_add(39)).map(|x| x.wrapping_add(40)).map(|x| x.wrapping_add(41)).map(|x| x.wrapping_add(42)).map(|x| x.wrapping_add(43)).map(|x| x.wrapping_add(44)).map(|x| x.wrapping_add(45)).map(|x| x.wrapping_add(46)).map(|x| x.wrapping_add(47)).map(|x| x.wrapping_add(48)).map(|x| x.wrapping_add(49)).map(|x| x.wrapping_add(50)).map(|x| x.wrapping_add(51)).map(|x| x.wrapping_add(52)).map(|x| x.wrapping_add(53)).map(|x| x.wrapping_add(54)).map(|x| x.wrapping_add(55)).map(|x| x.wrapping_add(56)).map(|x| x.wrapping_add(57)).map(|x| x.wrapping_add(58)).map(|x| x.wrapping_add(59)).map(|x| x.wrapping_add(60)).map(|x| x.wrapping_add(61)).map(|x| x.wrapping_add(62)).map(|x| x.wrapping_add(63)).map(|x| x.wrapping_add(64)).map(|x| x.wrapping_add(65)).map(|x| x.wrapping_add(66)).map(|x| x.wrapping_add(67)).map(|x| x.wrapping_add(68)).map(|x| x.wrapping_add(69)).map(|x| x.wrapping_add(70)).map(|x| x.wrapping_add(71)).map(|x| x.wrapping_add(72)).map(|x| x.wrapping_add(73)).map(|x| x.wrapping_add(74)).map(|x| x.wrapping_add(75)).map(|x| x.wrapping_add(76)).map(|x| x.wrapping_add(77)).map(|x| x.wrapping_add(78)).map(|x| x.wrapping_add(79)).map(|x| x.wrapping_add(80)).map(|x| x.wrapping_add(81)).map(|x| x.wrapping_add(82)).map(|x| x.wrapping_add(83)).map(|x| x.wrapping_add(84)).map(|x| x.wrapping_add(85)).map(|x| x.wrapping_add(86)).map(|x| x.wrapping_add(87)).map(|x| x.wrapping_add(88)).map(|x| x.wrapping_add(89)).map(|x| x.wrapping_add(90)).map(|x| x.wrapping_add(91)).map(|x| x.wrapping_add(92)).map(|x| x.wrapping_add(93)).map(|x| x.wrapping_add(94)).map(|x| x.wrapping_add(95)).map(|x| x.wrapping_add(96)).map(|x| x.wrapping_add(97)).map(|x| x.wrapping_add(98)).map(|x| x.wrapping_add(99)).map(|x| x.wrapping_add(100)).map(|x| x.wrapping_add(101)).map(|x| x.wrapping_add(102)).map(|x| x.wrapping_add(103)).map(|x| x.wrapping_add(104)).map(|x| x.wrapping_add(105)).map(|x| x.wrapping_add(106)).map(|x| x.wrapping_add(107)).map(|x| x.wrapping_add(108)).map(|x| x.wrapping_add(109)).map(|x| x.wrapping_add(110)).map(|x| x.wrapping_add(111)).map(|x| x.wrapping_add(112)).map(|x| x.wrapping_add(113)).map(|x| x.wrapping_add(114)).map(|x| x.wrapping_add(115)).map(|x| x.wrapping_add(116)).map(|x| x.wrapping_add(117)).map(|x| x.wrapping_add(118)).map(|x| x.wrapping_add(119)).map(|x| x.wrapping_add(120)).map(|x| x.wrapping_add(121)).map(|x| x.wrapping_add(122)).map(|x| x.wrapping_add(123)).map(|x| x.wrapping_add(124)).map(|x| x.wrapping_add(125)).map(|x| x.wrapping_add(126)).map(|x| x.wrapping_add(127)).map(|x| x.wrapping_add(128)).map(|x| x.wrapping_add(129)).map(|x| x.wrapping_add(130)).map(|x| x.wrapping_add(131)).map(|x| x.wrapping_add(132)).map(|x| x.wrapping_add(133)).map(|x| x.wrapping_add(134)).map(|x| x.wrapping_add(135)).map(|x| x.wrapping_add(136)).map(|x| x.wrapping_add(137)).map(|x| x.wrapping_add(138)).map(|x| x.wrapping_add(139)).map(|x| x.wrapping_add(140)).map(|x| x.wrapping_add(141)).map(|x| x.wrapping_add(142)).map(|x| x.wrapping_add(143)).map(|x| x.wrapping_add(144)).map(|x| x.wrapping_add(145)).map(|x| x.wrapping_add(146)).map(|x| x.wrapping_add(147)).map(|x| x.wrapping_add(148)).map(|x| x.wrapping_add(149)).map(|x| x.wrapping_add(150)).map(|x| x.wrapping_add(151)).map(|x| x.wrapping_add(152)).map(|x| x.wrapping_add(153)).map(|x| x.wrapping_add(154)).map(|x| x.wrapping_add(155)).map(|x| x.wrapping_add(156)).map(|x| x.wrapping_add(157)).map(|x| x.wrapping_add(158)).map(|x| x.wrapping_add(159)).map(|x| x.wrapping_add(160)).map(|x| x.wrapping_add(161)).map(|x| x.wrapping_add(162)).map(|x| x.wrapping_add(163)).map(|x| x.wrapping_add(164)).map(|x| x.wrapping_add(165)).map(|x| x.wrapping_add(166)).map(|x| x.wrapping_add(167)).map(|x| x.wrapping_add(168)).map(|x| x.wrapping_add(169)).map(|x| x.wrapping_add(170)).map(|x| x.wrapping_add(171)).map(|x| x.wrapping_add(172)).map(|x| x.wrapping_add(173)).map(|x| x.wrapping_add(174)).map(|x| x.wrapping_add(175)).map(|x| x.wrapping_add(176)).map(|x| x.wrapping_add(177)).map(|x| x.wrapping_add(178)).map(|x| x.wrapping_add(179)).map(|x| x.wrapping_add(180)).map(|x| x.wrapping_add(181)).map(|x| x.wrapping_add(182)).map(|x| x.wrapping_add(183)).map(|x| x.wrapping_add(184)).map(|x| x.wrapping_add(185)).map(|x| x.wrapping_add(186)).map(|x| x.wrapping_add(187)).map(|x| x.wrapping_add(188)).map(|x| x.wrapping_add(189)).map(|x| x.wrapping_add(190)).map(|x| x.wrapping_add(191)).map(|x| x.wrapping_add(192)).map(|x| x.wrapping_add(193)).map(|x| x.wrapping_add(194)).map(|x| x.wrapping_add(195)).map(|x| x.wrapping_add(196)).map(|x| x.wrapping_add(197)).map(|x| x.wrapping_add(198)).map(|x| x.wrapping_add(199)).sum(); println!("{}", s); } From c91ebda061183bd81d1351d5124a859213f7a05c Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Sat, 10 Oct 2026 08:27:27 +0000 Subject: [PATCH 23/23] flag-walk, flag-fuzz, coverage-flags: --rows must be lo:hi (a malformed value ran every row; Python raised) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01QXiEXbESemwqMLYKaWLDbT --- crates/mirth-lab/src/tools/coverage_flags.rs | 2 +- crates/mirth-lab/src/tools/flag_fuzz.rs | 2 +- crates/mirth-lab/src/tools/flag_walk.rs | 21 ++++++++++---------- 3 files changed, 13 insertions(+), 12 deletions(-) diff --git a/crates/mirth-lab/src/tools/coverage_flags.rs b/crates/mirth-lab/src/tools/coverage_flags.rs index f6a091e..1feb707 100644 --- a/crates/mirth-lab/src/tools/coverage_flags.rs +++ b/crates/mirth-lab/src/tools/coverage_flags.rs @@ -104,7 +104,7 @@ pub fn run(args: Args) -> anyhow::Result { let out = std::fs::canonicalize(&args.out)?; let opts = flag_model::options(&args.flags)?; let rows = flag_model::table(&args.table)?; - let idx = slice(&args.rows, rows.len()); + let idx = slice(&args.rows, rows.len())?; // Rows in order, a worker taking the next one when it is free. let next = AtomicUsize::new(0); std::thread::scope(|s| { diff --git a/crates/mirth-lab/src/tools/flag_fuzz.rs b/crates/mirth-lab/src/tools/flag_fuzz.rs index 9de92b5..01f17c1 100644 --- a/crates/mirth-lab/src/tools/flag_fuzz.rs +++ b/crates/mirth-lab/src/tools/flag_fuzz.rs @@ -57,7 +57,7 @@ pub fn run(args: Args) -> anyhow::Result { .filter_map(|l| serde_json::from_str::(l).ok()?["row"].as_u64().map(|r| r as usize)) .collect(); let me = std::env::current_exe()?; - for i in slice(&args.rows, rows.len()) { + for i in slice(&args.rows, rows.len())? { if done.contains(&i) { continue; } diff --git a/crates/mirth-lab/src/tools/flag_walk.rs b/crates/mirth-lab/src/tools/flag_walk.rs index 2b5e7bf..b47a7e3 100644 --- a/crates/mirth-lab/src/tools/flag_walk.rs +++ b/crates/mirth-lab/src/tools/flag_walk.rs @@ -93,19 +93,20 @@ pub fn copy_fixture(from: &Path, to: &Path) -> std::io::Result<()> { } /// Python's slice `a:b` of 0..n. -pub fn slice(spec: &str, n: usize) -> Vec { +/// Rows `lo:hi` (Python slice bounds, either may be empty or negative); empty means all. +pub fn slice(spec: &str, n: usize) -> anyhow::Result> { if spec.is_empty() { - return (0..n).collect(); + return Ok((0..n).collect()); } - let (lo, hi) = spec.split_once(':').unwrap_or((spec, "")); - let at = |s: &str, default: usize| -> usize { - match s.parse::() { - Ok(v) if v < 0 => (n as i64 + v).max(0) as usize, - Ok(v) => (v as usize).min(n), - Err(_) => default, + let Some((lo, hi)) = spec.split_once(':') else { anyhow::bail!("--rows takes lo:hi, not {spec:?}") }; + let at = |s: &str, default: usize| -> anyhow::Result { + if s.is_empty() { + return Ok(default); } + let v: i64 = s.parse().map_err(|_| anyhow::anyhow!("--rows takes lo:hi, not {spec:?}"))?; + Ok(if v < 0 { (n as i64 + v).max(0) as usize } else { (v as usize).min(n) }) }; - (at(lo, 0)..at(hi, n)).collect() + Ok((at(lo, 0)?..at(hi, n)?).collect()) } /// Python's repr of a string, as the findings have always shown texts. @@ -495,7 +496,7 @@ pub fn run(args: Args) -> anyhow::Result { let fixture = std::fs::canonicalize(&args.fixture)?; let _ = std::fs::remove_file(work.join("PAUSED")); let rows = flag_model::table(&args.table)?; - let idx = slice(&args.rows, rows.len()); + let idx = slice(&args.rows, rows.len())?; // Resume: rows with a result are done, except, with --recheck, those with findings, which // run first. let done = latest(&work);