From 3d31e561f618435a294e7e1ae592f625ed59161d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ralph=20K=C3=BCpper?= Date: Fri, 25 Sep 2026 14:25:44 +0200 Subject: [PATCH] perf(codegen): keep a unit's ordinary functions on the optimized machine pipeline when one function is over the fast-emit budget A function over PERRY_LL_FAST_EMIT_MAX_INSTRS used to switch its whole codegen unit to LLVM's O0 machine pipeline. It now takes, in order: a shadow-frame re-lowering for statepoint functions (optimized pipeline kept), containment in a module of its own emitted with FastISel on the optimized pipeline (O0 only past four times the budget), and whole-unit bounded emission only where the unit cannot be split. The compile prints a one-line machine-tier census when anything leaves the optimized tier. --- ...1321-fast-emit-per-function-containment.md | 8 + .../src/codegen/spec_preserve_none_tests.rs | 2 +- crates/perry-codegen/src/inprocess.rs | 165 +++++- .../src/inprocess/fast_emit_split.rs | 464 +++++++++++++++ .../src/inprocess/fast_emit_split_tests.rs | 551 ++++++++++++++++++ .../src/inprocess/optimize_emit.rs | 320 +++++----- crates/perry-codegen/src/lib.rs | 1 + crates/perry-codegen/src/linker.rs | 45 +- crates/perry-codegen/src/machine_tiers.rs | 136 +++++ crates/perry-codegen/src/native_emit.rs | 49 +- .../src/native_root_coverage/mod.rs | 3 +- .../perry/src/commands/compile/build_cache.rs | 3 + .../src/commands/compile/run_pipeline.rs | 6 + 13 files changed, 1561 insertions(+), 192 deletions(-) create mode 100644 changelog.d/11321-fast-emit-per-function-containment.md create mode 100644 crates/perry-codegen/src/inprocess/fast_emit_split.rs create mode 100644 crates/perry-codegen/src/inprocess/fast_emit_split_tests.rs create mode 100644 crates/perry-codegen/src/machine_tiers.rs diff --git a/changelog.d/11321-fast-emit-per-function-containment.md b/changelog.d/11321-fast-emit-per-function-containment.md new file mode 100644 index 0000000000..6c8dd78ccc --- /dev/null +++ b/changelog.d/11321-fast-emit-per-function-containment.md @@ -0,0 +1,8 @@ +- **codegen: a function over the optimized machine-pipeline budget no longer demotes its whole codegen unit.** `PERRY_LL_FAST_EMIT_MAX_INSTRS` (600k on x86-64, 100k elsewhere) used to switch the *unit's* `TargetMachine` to LLVM's O0 machine pipeline, because the optimization level is a per-module property. On the Claude Code bundle with #11179, one ~976k-instruction factory closure (89 % `gc.relocate`) took ~950 ordinary sibling functions to O0 with it; with this change it is re-lowered to 48,418 instructions and nothing in the bundle leaves the optimized machine pipeline. A function over the budget now takes, in order: + 1. **re-lowered**: a statepoint function is sent back to codegen through the existing typed RS4GC retry (`Rs4gcBudgetCause::MachineBudget`), lowered with its GC roots in a shadow frame, and its unit is compiled again at the same level through the *optimized* machine pipeline. Most of such a function is `gc.relocate` fan-out, which the shadow frame does not have. + 2. **contained**: a function still over the budget is moved into a module of its own after the IR pipeline (`inprocess/fast_emit_split.rs`) and partially linked back with `ld -r` the way codegen units already are. It is emitted by the optimized machine pipeline with **FastISel** instruction selection (O0 only past four times the budget): on the cc closure that is +2.6 % code on x86-64 (+1.3 % arm64) where O0 was ×23 (×49), at 55 s instead of 94 s of `llc` CPU. Every other function in the unit keeps the optimized machine pipeline. Locals the cut severs (internal functions, private constants, module globals) are promoted to external **hidden** symbols under a name made unique by a hash of the unit's function set. Each half carries its own stack map. + 3. **whole unit**: only where the unit cannot be split (COFF targets, or a host that cannot partially link the target's objects, e.g. a cross-architecture ELF target) does the whole unit take the bounded machine (same tier rule), which is the old behaviour. An alias/ifunc in the unit or a moved function in a comdat also declines the split, with a log line. +- The compile prints one `perry: machine code: …` summary line when anything left the optimized tier (`perry_codegen::machine_tiers`), so a regression is visible without per-unit logs. +- `PERRY_LL_FAST_EMIT_DUMP=` (diagnostic, not a cache input) keeps the bitcode of each contained module, and of a unit before a machine-budget re-lowering, for `llc` study. +- `optimize_and_emit_module*` now return one emission per part; `linker::finish_native_emission` finishes and joins them. +- Validation: `cargo test --release -p perry-codegen` (lib + all `tests/` suites) green on Linux x86-64; every `test_gap_*.ts` (1010 files) compiles to byte-identical objects on main and on this branch, so no function in the gap corpus reaches the budget. Sabotage: disabling the split, dropping the promotion, leaving the moved body on the sibling side, and disabling the re-lowering each turn their tests red. diff --git a/crates/perry-codegen/src/codegen/spec_preserve_none_tests.rs b/crates/perry-codegen/src/codegen/spec_preserve_none_tests.rs index b576baee6f..473e56e679 100644 --- a/crates/perry-codegen/src/codegen/spec_preserve_none_tests.rs +++ b/crates/perry-codegen/src/codegen/spec_preserve_none_tests.rs @@ -435,7 +435,7 @@ fn the_clone_entry_is_shrink_wrapped_frameless() { crate::codegen::helpers::native_stack_roots_enabled(), ) .expect("in-process -O3 -S pipeline"); - let asm = String::from_utf8(asm_bytes).expect("assembly is UTF-8"); + let asm = String::from_utf8(asm_bytes.concat()).expect("assembly is UTF-8"); // The label line for the clone (Mach-O prefixes `_`; ELF does not). let mut lines = asm.lines(); diff --git a/crates/perry-codegen/src/inprocess.rs b/crates/perry-codegen/src/inprocess.rs index f8fe02b633..f8165f5c7c 100644 --- a/crates/perry-codegen/src/inprocess.rs +++ b/crates/perry-codegen/src/inprocess.rs @@ -17,6 +17,7 @@ //! IR and flags this pipeline produces objects byte-identical to Homebrew //! clang 22's `clang -c`. +mod fast_emit_split; mod optimize_emit; use optimize_emit::optimize_and_emit; @@ -168,7 +169,7 @@ pub fn compile_ll_to_object_inprocess( clang_style_args: &[String], module_name: &str, native_roots: bool, -) -> Result> { +) -> Result>> { let (opt, mcpu_native, explicit_cpu, mllvm, emit_asm) = interpret_plan_args(clang_style_args)?; // Same guard as the external `opt` path (`linker::rs4gc_funclet_refusal`): // rewrite-statepoints-for-gc crashes on WinEH funclet pads, and here the @@ -295,12 +296,17 @@ pub(crate) fn parse_ir_text<'ctx>( /// Interpret plan argv (same grammar as `compile_ll_to_object_inprocess`) and /// run verify -> pass pipeline -> object emission on an already-built module. /// The native construction path calls this directly. +/// +/// Returns one emission per part: normally one, two when fast-emit +/// containment moved over-budget functions into a module of their own (see +/// `fast_emit_split`). `linker::finish_native_emission` turns the parts into +/// one object. pub(crate) fn optimize_and_emit_module( module: &inkwell::module::Module<'_>, effective_target: &str, clang_style_args: &[String], native_roots: bool, -) -> Result> { +) -> Result>> { optimize_and_emit_module_with_stats( module, effective_target, @@ -319,7 +325,7 @@ pub(crate) fn optimize_and_emit_module_with_stats( clang_style_args: &[String], native_roots: bool, stats: Option<&mut UnitCodegenStats>, -) -> Result> { +) -> Result>> { let (opt, mcpu_native, explicit_cpu, mllvm, emit_asm) = interpret_plan_args(clang_style_args)?; optimize_and_emit( module, @@ -398,18 +404,32 @@ fn module_instruction_census( /// Per-function instruction ceiling for LLVM's optimized machine pipeline. /// /// This budget is checked *after* the requested `default` IR pipeline has -/// completed. It changes neither JS lowering nor middle-end optimization; it -/// only asks the target machine to use its O0 instruction-selection, -/// live-interval and register-allocation pipeline for a unit containing an -/// extreme generated function. +/// completed. It changes neither JS lowering nor middle-end optimization. A +/// function over it takes, in order (see `crate::machine_tiers`): +/// +/// 1. **Re-lowered**: a statepoint function goes back to codegen, is lowered +/// with its GC roots in a shadow frame, and its unit is compiled again at +/// the same level through the *optimized* machine pipeline. Such a +/// function's size is mostly RS4GC relocation fan-out (163,100 of the +/// 227,108 instructions in the `@babel/parser` closure below are +/// `gc.relocate`), and that fan-out is what the machine pipeline is +/// super-linear in. +/// 2. **Contained**: a function still over the budget (or one that never had +/// statepoints) is moved into a module of its own (`fast_emit_split`) and +/// only it is emitted through LLVM's O0 machine pipeline. Every other +/// function of its unit keeps the optimized one. +/// 3. **Whole unit**: only where the unit cannot be split — COFF, or a host +/// that cannot partially link the target's objects — the whole unit takes +/// the O0 machine pipeline, which is the behaviour before containment. /// -/// **The demotion is a whole-unit act, so the budget must not be set where -/// ordinary functions pay for it.** A `TargetMachine`'s optimization level is -/// a per-module property: LLVM has no per-function escape from the optimized -/// machine pipeline (`optnone` reaches instruction selection and the optional -/// machine passes, but *not* LiveIntervals or the greedy register allocator — -/// measured below), so every ordinary function sharing the unit with one -/// extreme function is emitted through the O0 machine pipeline too. +/// Why the budget exists at all: a `TargetMachine`'s optimization level is a +/// per-module property, and LLVM has no per-function escape from the +/// optimized machine pipeline (`optnone` reaches instruction selection and +/// the optional machine passes, but *not* LiveIntervals or the greedy +/// register allocator — measured below). Before containment (tier 2) every +/// ordinary function sharing the unit with one extreme function was emitted +/// through the O0 machine pipeline too, which is why the numbers below were +/// taken per unit. /// /// Measured on `@babel/parser`'s unit 0, LLVM 22 / x86-64 / `-Os` IR pipeline: /// one 227,108-instruction closure (163,100 of those are `gc.relocate`) and @@ -422,13 +442,15 @@ fn module_instruction_census( /// | `optnone` on the closure only | 2,070,326 B | 621,693 B | 1.382 MiB | 9.5 s | 518 MiB | /// | the same unit *without* the closure | 1,448,633 B | — | 1.382 MiB | 6.4 s | 208 MiB | /// -/// So the siblings are pure loss: the fallback costs them 2.06 MiB of machine -/// code (168 of 282 functions change) to save ~6 s, and their emitted code is -/// byte-for-byte what a unit without the extreme function produces as soon as -/// the unit keeps the optimized pipeline. The `optnone` row is why this is a -/// budget and not a per-function demotion: it frees the siblings but bounds -/// neither time (9.5 s of 10.0 s) nor memory (518 MiB — *above* the -O2 arm), -/// because the greedy allocator still runs on the demoted function. +/// So the siblings were pure loss: the whole-unit fallback cost them 2.06 MiB +/// of machine code (168 of 282 functions change) to save ~6 s, and their +/// emitted code is byte-for-byte what a unit without the extreme function +/// produces as soon as the unit keeps the optimized pipeline — which is what +/// containment now gives them. The `optnone` row is why containment moves +/// the function into its own module instead of stamping it: `optnone` frees +/// the siblings but bounds neither time (9.5 s of 10.0 s) nor memory +/// (518 MiB — *above* the -O2 arm), because the greedy allocator still runs +/// on the demoted function. /// /// On x86-64 the ceiling is therefore set above the whole measured /// population of extreme generated functions rather than immediately below @@ -525,7 +547,7 @@ thread_local! { /// Thread-local budget seam; mutating the process environment would race the /// other LLVM tests in this binary. #[cfg(test)] -fn with_test_fast_emit_budget(cap: usize, run: impl FnOnce() -> T) -> T { +pub(crate) fn with_test_fast_emit_budget(cap: usize, run: impl FnOnce() -> T) -> T { with_test_fast_emit_budget_value(FastEmitBudget::Cap(cap), run) } @@ -544,6 +566,61 @@ fn with_test_fast_emit_budget_value(budget: FastEmitBudget, run: impl FnOnce( run() } +/// How far past the budget a function may be and still keep the optimized +/// machine pipeline with FastISel instruction selection ([`MachineTier`]). +const FAST_ISEL_TIER_FACTOR: usize = 4; + +/// The bounded machine configuration an over-budget function is emitted with. +/// +/// Measured on the Claude Code bundle's 975,886-instruction factory closure +/// (#11179; 870,626 of those instructions are `gc.relocate`), `llc` 22 on the +/// post-`default` IR, one sample each, `.text` of the function alone: +/// +/// | machine configuration | x86-64 `.text` | x86-64 CPU | x86-64 RSS | arm64 `.text` | arm64 CPU | arm64 RSS | +/// |---|---|---|---|---|---|---| +/// | optimized (`-O2`) | 280,366 B | 93.8 s | 1.22 GB | 223,568 B | 59.8 s | 1.49 GB | +/// | optimized + FastISel (this tier) | 287,654 B | 55.1 s | 1.21 GB | 226,512 B | 56.3 s | 1.49 GB | +/// | O0 (the old fallback) | 6,584,974 B | 20.1 s | 1.28 GB | 10,952,192 B | 12.6 s | 1.65 GB | +/// +/// On x86-64 95 % of the optimized pipeline's time on that function is +/// SelectionDAG instruction selection (84.5 of 89.4 s under `-time-passes`; +/// the greedy allocator took 0.4 s), and FastISel is exactly the part of the +/// O0 pipeline that removes it. So this tier costs +2.6 % code where O0 costs +/// ×23 (×48 on arm64), and it is a per-`TargetMachine` switch, so it needs +/// no process-global `cl::opt`. +/// +/// O0 stays as the backstop only for a function more than +/// [`FAST_ISEL_TIER_FACTOR`] times over the budget, which is where the +/// measurements stop: the degradation grows with the excess instead of +/// stepping to ×23 at the budget. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MachineTier { + /// The requested optimization level with FastISel instruction selection: + /// the optimized register allocator and machine passes are kept. + FastIsel, + /// LLVM's O0 machine pipeline (FastISel + the fast register allocator). + O0, +} + +impl MachineTier { + pub(crate) fn for_function(instructions: usize, cap: usize) -> Self { + if instructions <= cap.saturating_mul(FAST_ISEL_TIER_FACTOR) { + MachineTier::FastIsel + } else { + MachineTier::O0 + } + } + + fn describe(self) -> &'static str { + match self { + MachineTier::FastIsel => { + "the optimized machine pipeline with FastISel instruction selection" + } + MachineTier::O0 => "LLVM's O0 machine pipeline", + } + } +} + /// One extreme function which selected bounded machine-code emission, and how /// many defined functions in its unit are demoted along with it. #[derive(Debug, Clone, PartialEq, Eq)] @@ -551,20 +628,39 @@ pub struct FastEmitFallback { pub name: String, pub instructions: usize, pub cap: usize, - /// Defined functions in the unit — the size of the collateral, since the - /// machine pipeline is selected per module and not per function. + /// Defined functions in the unit — the size of the collateral when the + /// unit could not be split, since the machine pipeline is selected per + /// module and not per function. pub unit_functions: usize, + /// Whether the over-budget functions were moved into a module of their + /// own (`fast_emit_split`), so only they took the bounded pipeline and + /// every other function in the unit kept the optimized one. + pub contained: bool, + /// The bounded machine configuration it was emitted with. + pub tier: MachineTier, } impl std::fmt::Display for FastEmitFallback { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + let how = self.tier.describe(); + if self.contained { + return write!( + f, + "`{}` has {} instructions after IR optimization, above the optimized \ + machine-pipeline budget {}; keeping the requested IR optimization, then \ + emitting this function alone through {how} to bound instruction selection. \ + Every function of its {}-function unit that is under the budget keeps the \ + optimized machine pipeline. Override with PERRY_LL_FAST_EMIT_MAX_INSTRS= \ + (raise) or =0 (disable).", + self.name, self.instructions, self.cap, self.unit_functions + ); + } write!( f, "`{}` has {} instructions after IR optimization, above the optimized machine-pipeline \ budget {}; keeping the requested IR optimization, then emitting this unit — all {} \ - of its defined functions, not only this one — through LLVM's O0 machine pipeline to \ - bound instruction selection, live intervals and register allocation. LLVM selects \ - that pipeline per module, so the siblings are demoted too and grow: shrinking this \ + of its defined functions, not only this one — through {how}. The unit could not \ + be split for this target, so the siblings take that pipeline too: shrinking this \ function is what lifts the whole unit back. Override with \ PERRY_LL_FAST_EMIT_MAX_INSTRS= (raise) or =0 (disable).", self.name, self.instructions, self.cap, self.unit_functions @@ -610,6 +706,8 @@ fn fast_emit_fallbacks( instructions, cap, unit_functions: defined, + contained: false, + tier: MachineTier::for_function(instructions, cap), }) .collect() } @@ -652,6 +750,12 @@ pub(crate) enum Rs4gcBudgetCause { /// RS4GC finished, but its relocation fan-out made the rewritten body too /// large for the normal optimization pipeline. PostRewrite { post_instructions: usize }, + /// The IR pipeline finished, but the optimized function is over the + /// machine-pipeline budget ([`DEFAULT_FAST_EMIT_MAX_INSTRS_X86_64`]). + /// Re-lowering it onto a shadow frame removes the statepoint relocations + /// that make up most of such a function, so it can keep the optimized + /// machine pipeline instead of the bounded one. + MachineBudget { instructions: usize }, } #[derive(Debug, Clone, PartialEq, Eq)] @@ -961,6 +1065,13 @@ fn rewrite_budget_message(violation: &Rs4gcBudgetViolation, retry: bool) -> Stri violation.name, violation.cap ) } + Rs4gcBudgetCause::MachineBudget { instructions } => format!( + "`{}` has {instructions} instructions after IR optimization, above the optimized \ + machine-pipeline budget {}; most of a statepoint function this size is relocation \ + fan-out, so {outcome}. Override with PERRY_LL_FAST_EMIT_MAX_INSTRS= (raise) or \ + =0 (disable).", + violation.name, violation.cap + ), } } diff --git a/crates/perry-codegen/src/inprocess/fast_emit_split.rs b/crates/perry-codegen/src/inprocess/fast_emit_split.rs new file mode 100644 index 0000000000..7f6187818e --- /dev/null +++ b/crates/perry-codegen/src/inprocess/fast_emit_split.rs @@ -0,0 +1,464 @@ +//! Per-function fast-emit containment. +//! +//! A `TargetMachine` selects its machine pipeline for a whole module, so the +//! bounded machine pipeline chosen for one extreme function (see +//! [`super::DEFAULT_FAST_EMIT_MAX_INSTRS_X86_64`]) used to reach every +//! ordinary function in its codegen unit too. On the Claude Code bundle one +//! 988k-instruction factory closure demoted 950 siblings. +//! +//! This module moves the over-budget functions into a module of their own +//! after the IR pipeline has run, so each half is emitted by its own target +//! machine: the siblings by the unit's optimized one, the extreme functions +//! by the bounded one. The two emissions become two objects that the caller +//! partially links (`linker::merge_unit_objects`, the step that already joins +//! codegen units), so the rest of the backend never sees two objects. +//! +//! What has to hold across the cut: +//! +//! * Every internal symbol that the moved functions reference (functions, +//! string constants, module globals) is defined on the sibling side and +//! declared on the contained side. It is promoted to external linkage with +//! hidden visibility under a name made unique by a hash of the unit's +//! function set, so no two units — and no two Perry modules — can define +//! the same promoted name. Hidden keeps it out of the final image's +//! dynamic symbol table and keeps its references PC-relative. +//! * A moved function that was internal is promoted the same way, because its +//! callers stay on the sibling side. +//! * Each object carries its own statepoint stack map for exactly the +//! functions it defines, and the partial link concatenates the compact +//! `.perry_gcmap` sections the way it already does for codegen units. +//! * Module-level `asm` (the Mach-O `.no_dead_strip` for the stack map) is +//! kept on both sides; `llvm.used`-style appending arrays and every global +//! initializer stay on the sibling side only. +//! +//! Shapes that this cut cannot express safely — an alias or ifunc anywhere in +//! the unit, a moved function in a comdat — decline containment, and the unit +//! falls back to whole-unit bounded emission exactly as before. + +use std::collections::HashSet; + +use inkwell::module::Module; +use llvm_sys::comdat::{LLVMGetComdat, LLVMSetComdat}; +use llvm_sys::core::*; +use llvm_sys::prelude::*; +use llvm_sys::{LLVMLinkage, LLVMTypeKind, LLVMVisibility}; + +/// Suffix infix of every promoted name, so a symbol table shows where it came +/// from. +pub(super) const PROMOTED_INFIX: &str = ".perry.fe."; + +/// Why a unit could not be split. The caller logs it and keeps the old +/// whole-unit behaviour. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(super) struct SplitDeclined(pub String); + +impl std::fmt::Display for SplitDeclined { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(&self.0) + } +} + +/// Whether this host can partially link objects for `effective_target`. +/// +/// The two halves are joined by `ld -r` (`linker::merge_unit_objects`). COFF +/// has no relocatable link (codegen units are archived there instead, and an +/// archive cannot nest inside another unit's archive), and a host linker +/// only reads its own object format — and, for GNU ld, its own architecture. +/// Everywhere else the unit keeps whole-unit bounded emission. +pub(super) fn split_emission_supported(effective_target: &str) -> bool { + let arch = effective_target + .split('-') + .next() + .unwrap_or(effective_target); + let apple = effective_target.contains("apple"); + if effective_target.contains("windows") { + return false; + } + if cfg!(target_os = "macos") { + return apple; + } + if cfg!(target_os = "linux") { + let host = std::env::consts::ARCH; + let same_arch = match arch { + "x86_64" | "x86_64h" | "amd64" => host == "x86_64", + "aarch64" | "arm64" => host == "aarch64", + _ => false, + }; + return !apple && same_arch; + } + false +} + +fn value_name(value: LLVMValueRef) -> String { + let mut len = 0usize; + let ptr = unsafe { LLVMGetValueName2(value, &mut len) }; + if ptr.is_null() || len == 0 { + return String::new(); + } + let bytes = unsafe { std::slice::from_raw_parts(ptr as *const u8, len) }; + String::from_utf8_lossy(bytes).into_owned() +} + +fn set_value_name(value: LLVMValueRef, name: &str) { + unsafe { LLVMSetValueName2(value, name.as_ptr() as *const _, name.len()) }; +} + +fn is_local(linkage: LLVMLinkage) -> bool { + matches!( + linkage, + LLVMLinkage::LLVMInternalLinkage | LLVMLinkage::LLVMPrivateLinkage + ) +} + +fn is_definition(global: LLVMValueRef) -> bool { + unsafe { LLVMIsDeclaration(global) == 0 } +} + +fn functions(module: LLVMModuleRef) -> Vec { + let mut out = Vec::new(); + let mut f = unsafe { LLVMGetFirstFunction(module) }; + while !f.is_null() { + out.push(f); + f = unsafe { LLVMGetNextFunction(f) }; + } + out +} + +fn global_variables(module: LLVMModuleRef) -> Vec { + let mut out = Vec::new(); + let mut g = unsafe { LLVMGetFirstGlobal(module) }; + while !g.is_null() { + out.push(g); + g = unsafe { LLVMGetNextGlobal(g) }; + } + out +} + +/// FNV-1a over every defined function name (module order) and the moved +/// set. Deterministic, so the object cache and reproducible builds see the +/// same promoted names every time; distinct per unit, because no two units +/// define the same set of functions. +fn unit_token(module: LLVMModuleRef, moved: &[LLVMValueRef]) -> u64 { + let mut hash: u64 = 0xcbf2_9ce4_8422_2325; + let mut eat = |bytes: &[u8]| { + for b in bytes.iter().chain(std::iter::once(&0u8)) { + hash ^= u64::from(*b); + hash = hash.wrapping_mul(0x0000_0100_0000_01b3); + } + }; + for f in functions(module) { + if is_definition(f) { + eat(value_name(f).as_bytes()); + } + } + eat(b"--moved--"); + for f in moved { + eat(value_name(*f).as_bytes()); + } + hash +} + +/// Every internal/private global value a moved function body references, +/// through instruction operands and nested constant expressions (not through +/// other globals' initializers, which stay on the sibling side). +fn locals_referenced_by(function: LLVMValueRef) -> Vec { + let mut seen_constants: HashSet = HashSet::new(); + let mut found: Vec = Vec::new(); + let mut found_set: HashSet = HashSet::new(); + let mut stack: Vec = Vec::new(); + + let mut visit_operand = |value: LLVMValueRef, stack: &mut Vec| { + if value.is_null() { + return; + } + unsafe { + if !LLVMIsAGlobalValue(value).is_null() { + if is_local(LLVMGetLinkage(value)) && found_set.insert(value) { + found.push(value); + } + } else if !LLVMIsAConstant(value).is_null() && seen_constants.insert(value) { + stack.push(value); + } + } + }; + + unsafe { + if LLVMHasPersonalityFn(function) != 0 { + visit_operand(LLVMGetPersonalityFn(function), &mut stack); + } + let mut bb = LLVMGetFirstBasicBlock(function); + while !bb.is_null() { + let mut inst = LLVMGetFirstInstruction(bb); + while !inst.is_null() { + let n = LLVMGetNumOperands(inst); + for i in 0..n.max(0) as u32 { + visit_operand(LLVMGetOperand(inst, i), &mut stack); + } + inst = LLVMGetNextInstruction(inst); + } + bb = LLVMGetNextBasicBlock(bb); + } + while let Some(constant) = stack.pop() { + let n = LLVMGetNumOperands(constant); + for i in 0..n.max(0) as u32 { + visit_operand(LLVMGetOperand(constant, i), &mut stack); + } + } + } + found +} + +/// Give a local global value external linkage, hidden visibility and a +/// unit-unique name. +fn promote(global: LLVMValueRef, token: u64, anon: &mut usize) { + let base = value_name(global); + let base = if base.is_empty() { + *anon += 1; + format!("anon{}", *anon) + } else { + base + }; + set_value_name(global, &format!("{base}{PROMOTED_INFIX}{token:016x}")); + unsafe { + LLVMSetLinkage(global, LLVMLinkage::LLVMExternalLinkage); + LLVMSetVisibility(global, LLVMVisibility::LLVMHiddenVisibility); + } +} + +/// Remove a function's body in place, keeping its type, +/// attributes, calling convention, GC strategy and visibility — the things a +/// caller's code generation reads from the callee. +/// +/// The C API has no `deleteBody`, so this is what `DeleteDeadBlocks` does: +/// replace every instruction's uses with poison, erase the instructions +/// (dropping their references to blocks and values), then delete the empty +/// blocks. The caller settles the linkage: a declaration may not be local. +pub(super) fn strip_body(function: LLVMValueRef) { + unsafe { + let mut bb = LLVMGetFirstBasicBlock(function); + while !bb.is_null() { + let mut inst = LLVMGetFirstInstruction(bb); + while !inst.is_null() { + let ty = LLVMTypeOf(inst); + if LLVMGetTypeKind(ty) != LLVMTypeKind::LLVMVoidTypeKind + && !LLVMGetFirstUse(inst).is_null() + { + LLVMReplaceAllUsesWith(inst, LLVMGetPoison(ty)); + } + inst = LLVMGetNextInstruction(inst); + } + bb = LLVMGetNextBasicBlock(bb); + } + let mut bb = LLVMGetFirstBasicBlock(function); + while !bb.is_null() { + let mut inst = LLVMGetLastInstruction(bb); + while !inst.is_null() { + let prev = LLVMGetPreviousInstruction(inst); + LLVMInstructionEraseFromParent(inst); + inst = prev; + } + bb = LLVMGetNextBasicBlock(bb); + } + loop { + let bb = LLVMGetFirstBasicBlock(function); + if bb.is_null() { + break; + } + LLVMDeleteBasicBlock(bb); + } + if LLVMHasPersonalityFn(function) != 0 { + LLVMSetPersonalityFn(function, std::ptr::null_mut()); + } + LLVMSetComdat(function, std::ptr::null_mut()); + } +} + +/// A declaration may only be external (or extern_weak). Visibility is kept: +/// a hidden declaration is what lets the backend address its definition in +/// the other half PC-relatively. +fn declaration_linkage(global: LLVMValueRef) { + unsafe { + if LLVMGetLinkage(global) != LLVMLinkage::LLVMExternalWeakLinkage { + LLVMSetLinkage(global, LLVMLinkage::LLVMExternalLinkage); + } + } +} + +/// Split `module` in place: after this returns `Ok(Some(contained))`, +/// `module` defines every function except `moved`, and `contained` defines +/// exactly `moved` and declares everything they reference. +/// +/// `Ok(None)`: every defined function in the unit is in `moved`, so there is +/// nothing to protect and the unit is emitted whole by the bounded machine. +/// +/// `Err` means nothing was cut and `module` is still one emittable unit (the +/// only mutation that may already have happened is the promotion of locals, +/// which is correct for a single object too). +pub(super) fn split_moved_functions<'ctx>( + module: &Module<'ctx>, + moved_names: &[String], +) -> Result>, SplitDeclined> { + #[cfg(test)] + if TEST_DECLINE_SPLIT.with(std::cell::Cell::get) { + return Err(SplitDeclined("declined by the test seam".into())); + } + let raw = module.as_mut_ptr(); + unsafe { + if !LLVMGetFirstGlobalAlias(raw).is_null() { + return Err(SplitDeclined("the unit defines a global alias".into())); + } + if !LLVMGetFirstGlobalIFunc(raw).is_null() { + return Err(SplitDeclined("the unit defines an ifunc".into())); + } + } + let mut moved: Vec = Vec::with_capacity(moved_names.len()); + for name in moved_names { + let f = module + .get_function(name) + .ok_or_else(|| SplitDeclined(format!("`{name}` is not in the unit")))?; + let f = inkwell::values::AsValueRef::as_value_ref(&f); + if !is_definition(f) { + return Err(SplitDeclined(format!("`{name}` has no body"))); + } + if unsafe { !LLVMGetComdat(f).is_null() } { + return Err(SplitDeclined(format!("`{name}` is in a comdat"))); + } + moved.push(f); + } + let moved_set: HashSet = moved.iter().copied().collect(); + let siblings = functions(raw) + .into_iter() + .filter(|f| is_definition(*f) && !moved_set.contains(f)) + .count(); + if siblings == 0 { + return Ok(None); + } + + // 1. Promote, in the original, every local the cut would sever. + let token = unit_token(raw, &moved); + let mut to_promote: Vec = Vec::new(); + let mut promote_set: HashSet = HashSet::new(); + for f in &moved { + if is_local(unsafe { LLVMGetLinkage(*f) }) && promote_set.insert(*f) { + to_promote.push(*f); + } + for local in locals_referenced_by(*f) { + if promote_set.insert(local) { + to_promote.push(local); + } + } + } + let mut anon = 0usize; + for global in &to_promote { + promote(*global, token, &mut anon); + } + let moved_names_promoted: Vec = moved.iter().map(|f| value_name(*f)).collect(); + + // 2. Clone; the clone becomes the contained module. + let contained = unsafe { Module::new(LLVMCloneModule(raw)) }; + let craw = contained.as_mut_ptr(); + let keep: HashSet = moved_names_promoted.iter().cloned().collect(); + unsafe { + for g in global_variables(craw) { + if LLVMGetLinkage(g) == LLVMLinkage::LLVMAppendingLinkage { + LLVMDeleteGlobal(g); + } + } + // Locals keep their (now invalid for a declaration) linkage for one + // more step: they are deleted below, never turned into external + // declarations of a symbol nobody exports. + for f in functions(craw) { + if is_definition(f) && !keep.contains(&value_name(f)) { + strip_body(f); + if !is_local(LLVMGetLinkage(f)) { + declaration_linkage(f); + } + } + } + for g in global_variables(craw) { + if is_definition(g) { + LLVMSetInitializer(g, std::ptr::null_mut()); + LLVMSetComdat(g, std::ptr::null_mut()); + if !is_local(LLVMGetLinkage(g)) { + declaration_linkage(g); + } + } + } + // What is left local was referenced only by the stripped bodies and + // initializers, so nothing uses it any more. + for f in functions(craw) { + if is_local(LLVMGetLinkage(f)) { + if !LLVMGetFirstUse(f).is_null() { + return Err(SplitDeclined(format!( + "local function `{}` is still referenced after the cut", + value_name(f) + ))); + } + LLVMDeleteFunction(f); + } + } + for g in global_variables(craw) { + if is_local(LLVMGetLinkage(g)) { + if !LLVMGetFirstUse(g).is_null() { + return Err(SplitDeclined(format!( + "local global `{}` is still referenced after the cut", + value_name(g) + ))); + } + LLVMDeleteGlobal(g); + } + } + } + // Drop every declaration nothing references any more (the unit's other + // functions and globals): they emit nothing, but they are most of the + // clone's symbol table. + unsafe { + for f in functions(craw) { + if !is_definition(f) && LLVMGetFirstUse(f).is_null() { + LLVMDeleteFunction(f); + } + } + for g in global_variables(craw) { + if !is_definition(g) && LLVMGetFirstUse(g).is_null() { + LLVMDeleteGlobal(g); + } + } + } + contained.verify().map_err(|e| { + SplitDeclined(format!( + "the contained module does not verify:\n{}", + e.to_string() + )) + })?; + + // 3. Only now cut the original: past this point there is no way back to + // one module, and a verifier failure is a Perry bug. + for f in &moved { + strip_body(*f); + declaration_linkage(*f); + } + Ok(Some(contained)) +} + +#[cfg(test)] +thread_local! { + static TEST_DECLINE_SPLIT: std::cell::Cell = const { std::cell::Cell::new(false) }; +} + +/// Test seam: run `body` with containment declined, i.e. the pre-containment +/// whole-unit behaviour — the arm that proves the containment tests can fail. +#[cfg(test)] +pub(super) fn with_split_declined(body: impl FnOnce() -> T) -> T { + struct Restore(bool); + impl Drop for Restore { + fn drop(&mut self) { + TEST_DECLINE_SPLIT.with(|cell| cell.set(self.0)); + } + } + let _restore = Restore(TEST_DECLINE_SPLIT.replace(true)); + body() +} + +#[cfg(test)] +#[path = "fast_emit_split_tests.rs"] +mod tests; diff --git a/crates/perry-codegen/src/inprocess/fast_emit_split_tests.rs b/crates/perry-codegen/src/inprocess/fast_emit_split_tests.rs new file mode 100644 index 0000000000..90c89b1ab8 --- /dev/null +++ b/crates/perry-codegen/src/inprocess/fast_emit_split_tests.rs @@ -0,0 +1,551 @@ +//! Tests for per-function fast-emit containment (`fast_emit_split`). + +use super::super::*; +use super::{split_emission_supported, with_split_declined, PROMOTED_INFIX}; + +fn host_target() -> String { + crate::codegen::default_target_triple() +} + +/// Whether this host can run the split tests. On the hosts CI and developers +/// use (macOS, Linux) the split MUST be supported for the host triple — +/// otherwise a regression in `split_emission_supported` would turn every +/// containment test below into a silent skip. +fn split_host() -> bool { + if cfg!(any(target_os = "macos", target_os = "linux")) { + assert!( + split_emission_supported(&host_target()), + "containment must be supported for the host triple {}", + host_target() + ); + true + } else { + false + } +} + +/// One function over the budget, one ordinary function that is nowhere +/// near it, and one external callee so nothing folds away. `narrow` +/// holds three values across three calls, which is what makes its +/// machine code differ between the optimized and the O0 register +/// allocators. +fn sibling_cost_fixture(with_wide: bool) -> String { + let wide = r#" +define i64 @wide(i64 %n) { +entry: + %a = call i64 @src(i64 %n) + %b = call i64 @src(i64 %a) + %c = call i64 @src(i64 %b) + %d = call i64 @src(i64 %c) + %s = add i64 %a, %b + %t = add i64 %s, %c + %u = add i64 %t, %d + ret i64 %u +} +"#; + format!( + r#" +declare i64 @src(i64) + +define i64 @narrow(i64 %x, i64 %y) {{ +entry: + %a = call i64 @src(i64 %x) + %b = call i64 @src(i64 %y) + %c = call i64 @src(i64 %a) + %s = add i64 %a, %b + %t = add i64 %s, %c + ret i64 %t +}} +{}"#, + if with_wide { wide } else { "" } + ) +} + +/// The assembly of one function, from its label to the end of its body. +/// Tolerates ELF (`narrow:` / `.size`) and Mach-O (`_narrow:`) spelling. +fn function_assembly(asm: &str, name: &str) -> String { + let label_elf = format!("{name}:"); + let label_macho = format!("_{name}:"); + let mut body: Vec<&str> = Vec::new(); + let mut inside = false; + for line in asm.lines() { + let trimmed = line.trim(); + if !inside { + inside = trimmed == label_elf || trimmed == label_macho; + continue; + } + let next_symbol = trimmed.ends_with(':') + && !trimmed.starts_with('.') + && !trimmed.starts_with('L') + && !trimmed.contains(' '); + if trimmed.starts_with(".size") || trimmed == ".cfi_endproc" || next_symbol { + break; + } + body.push(line); + } + assert!( + body.len() > 3, + "no body extracted for `{name}` — the assertion below would be vacuous:\n{asm}" + ); + body.join("\n") +} + +/// Emit `ir` at `-Os -S` under `budget`; one assembly text per emitted part +/// (two when containment split the unit), the functions over the budget, +/// and whether they were contained. +fn emit_assembly( + ir: &str, + module_name: &str, + budget: FastEmitBudget, +) -> (Vec, Vec, bool) { + global_init(&[]); + let target = crate::codegen::default_target_triple(); + let context = Context::create(); + let module = parse_ir_text(&context, ir, module_name).expect("fixture parses"); + let mut stats = UnitCodegenStats::default(); + let asm = with_test_fast_emit_budget_value(budget, || { + optimize_and_emit_module_with_stats( + &module, + &target, + &["-Os".into(), "-S".into()], + false, + Some(&mut stats), + ) + }) + .expect("the fixture emits"); + ( + asm.into_iter() + .map(|part| String::from_utf8(part).expect("LLVM emits UTF-8 assembly")) + .collect(), + stats + .fast_emit_fallbacks + .iter() + .map(|f| f.name.clone()) + .collect(), + stats.fast_emit_fallbacks.iter().all(|f| f.contained), + ) +} + +fn defines(asm: &str, name: &str) -> bool { + asm.lines().any(|line| { + let t = line.trim(); + t == format!("{name}:") || t == format!("_{name}:") + }) +} + +/// The ordinary function keeps the optimized machine pipeline when an +/// extreme function shares its unit: its machine code is byte-for-byte what +/// the unit emits with no budget at all, and what a unit *without* the +/// extreme function emits. The extreme function is emitted alone in the +/// second part. +#[test] +fn containment_keeps_ordinary_siblings_on_the_optimized_pipeline() { + if !split_host() { + return; + } + let with_wide = sibling_cost_fixture(true); + let alone = sibling_cost_fixture(false); + + let (undemoted, none, _) = emit_assembly(&with_wide, "sibling_cost_ok", FastEmitBudget::Off); + assert!(none.is_empty(), "this arm must not demote: {none:?}"); + assert_eq!(undemoted.len(), 1, "no budget, no split"); + let (solo, _, _) = emit_assembly(&alone, "sibling_cost_alone", FastEmitBudget::Off); + assert_eq!( + function_assembly(&undemoted[0], "narrow"), + function_assembly(&solo[0], "narrow"), + "an undemoted unit emits an ordinary function exactly as a unit without the extreme \ + function does" + ); + + let (parts, over, contained) = + emit_assembly(&with_wide, "sibling_cost_contained", FastEmitBudget::Cap(7)); + assert_eq!(over, ["wide"], "only `wide` is over the budget"); + assert!(contained, "the stats must report the containment"); + assert_eq!(parts.len(), 2, "the unit must be emitted in two parts"); + assert!(defines(&parts[0], "narrow") && !defines(&parts[0], "wide")); + assert!(defines(&parts[1], "wide") && !defines(&parts[1], "narrow")); + assert_eq!( + function_assembly(&parts[0], "narrow"), + function_assembly(&undemoted[0], "narrow"), + "`narrow` is under the budget and must keep the optimized machine pipeline" + ); +} + +/// The discriminating arm for the test above: with the split declined and +/// the O0 machine (the pre-containment behaviour), the same `narrow` IS +/// compiled differently. +/// If this ever stops failing to match, the fixture no longer tells the two +/// machine pipelines apart and the containment test is vacuous. +#[test] +fn without_containment_the_sibling_is_demoted() { + let with_wide = sibling_cost_fixture(true); + let (undemoted, _, _) = emit_assembly(&with_wide, "sibling_cost_ok2", FastEmitBudget::Off); + // Cap 1: both functions are far past four times the budget, so the + // declined unit takes the O0 backstop — the machine whose effect on an + // ordinary sibling this arm demonstrates. + let (demoted, over, contained) = with_split_declined(|| { + emit_assembly(&with_wide, "sibling_cost_demoted", FastEmitBudget::Cap(1)) + }); + assert_eq!(over, ["wide", "narrow"]); + assert!(!contained); + assert_eq!(demoted.len(), 1, "a declined split emits one part"); + assert_ne!( + function_assembly(&demoted[0], "narrow"), + function_assembly(&undemoted[0], "narrow"), + "whole-unit demotion must reach `narrow`, or this fixture cannot tell the pipelines apart" + ); +} + +/// Every edge the cut severs, in one program that is linked and run: an +/// internal extreme function called directly and through a global table, an +/// internal helper it calls back across the cut, a private string constant +/// and an internal mutable global it shares with the sibling side. A missing +/// promotion is an undefined symbol at link time; a wrong one (two copies of +/// the global) changes the exit code. +fn cross_cut_program() -> &'static str { + r#" +@.msg = private unnamed_addr constant [4 x i8] c"abc\00" +@counter = internal global i64 0 +@table = internal global [1 x ptr] [ptr @wide] + +define internal i64 @helper(i64 %x) noinline { +entry: + %c = load i64, ptr @counter + %c1 = add i64 %c, 1 + store i64 %c1, ptr @counter + %r = add i64 %x, %c1 + ret i64 %r +} + +define internal i64 @wide(i64 %n) noinline { +entry: + %a = call i64 @helper(i64 %n) + %b = call i64 @helper(i64 %a) + %c0 = call i64 @helper(i64 %b) + %c1 = call i64 @helper(i64 %c0) + %c2 = call i64 @helper(i64 %c1) + %c3 = call i64 @helper(i64 %c2) + %c4 = call i64 @helper(i64 %c3) + %c5 = call i64 @helper(i64 %c4) + %c = call i64 @helper(i64 %c5) + %p = getelementptr inbounds [4 x i8], ptr @.msg, i64 0, i64 1 + %ch = load i8, ptr %p + %chw = zext i8 %ch to i64 + %s = add i64 %c, %chw + %k = load i64, ptr @counter + %t = mul i64 %s, %k + %u = xor i64 %t, %a + %v = sub i64 %u, %b + ret i64 %v +} + +define i32 @main(i32 %argc, ptr %argv) { +entry: + %seed = sext i32 %argc to i64 + store i64 %seed, ptr @counter + %f = load ptr, ptr @table + %x = call i64 %f(i64 %seed) + %y = call i64 @wide(i64 %seed) + %k = load i64, ptr @counter + %sum = add i64 %x, %y + %all = add i64 %sum, %k + %m = urem i64 %all, 251 + %r = trunc i64 %m to i32 + ret i32 %r +} +"# +} + +fn link_and_run(object: &[u8], label: &str) -> i32 { + let dir = std::env::temp_dir().join(format!( + "perry_fast_emit_split_{label}_{}", + std::process::id() + )); + std::fs::create_dir_all(&dir).expect("scratch dir"); + let obj = dir.join("prog.o"); + let exe = dir.join("prog"); + std::fs::write(&obj, object).expect("write object"); + let cc = crate::linker::find_clang().expect("a C compiler driver links the fixture"); + let linked = std::process::Command::new(cc) + .arg(&obj) + .arg("-o") + .arg(&exe) + .output() + .expect("run the linker"); + assert!( + linked.status.success(), + "{label}: link failed:\n{}", + String::from_utf8_lossy(&linked.stderr) + ); + let status = std::process::Command::new(&exe) + .status() + .expect("run the fixture"); + let _ = std::fs::remove_dir_all(&dir); + status.code().expect("the fixture exits normally") +} + +fn emit_program( + budget: FastEmitBudget, + label: &str, +) -> (Vec>, Vec, Vec) { + global_init(&[]); + let target = host_target(); + let args: Vec = vec!["-Os".into(), "-c".into()]; + let context = Context::create(); + let module = parse_ir_text(&context, cross_cut_program(), label).expect("fixture parses"); + let mut stats = UnitCodegenStats::default(); + let parts = with_test_fast_emit_budget_value(budget, || { + optimize_and_emit_module_with_stats(&module, &target, &args, false, Some(&mut stats)) + }) + .expect("the fixture emits"); + let object = + crate::linker::finish_native_emission(parts.clone(), &target, &args).expect("finishes"); + (parts, object, stats.fast_emit_fallbacks) +} + +#[test] +fn containment_preserves_every_edge_across_the_cut() { + if !split_host() { + return; + } + let (whole, whole_object, _) = emit_program(FastEmitBudget::Off, "whole"); + assert_eq!(whole.len(), 1); + let expected = link_and_run(&whole_object, "whole"); + + let (parts, split_object, over) = emit_program(FastEmitBudget::Cap(12), "split"); + assert!( + over.len() == 1 && over[0].name == "wide" && over[0].contained, + "`wide` alone must be over the budget and contained: {over:?}" + ); + assert_eq!(parts.len(), 2, "the unit must be emitted in two parts"); + assert_eq!( + link_and_run(&split_object, "split"), + expected, + "the split program must compute what the whole one does" + ); +} + +/// The promoted names are hidden, unit-unique, and only the severed locals +/// are promoted. +#[test] +fn promotion_is_hidden_unique_and_minimal() { + if !split_host() { + return; + } + let context = Context::create(); + let module = parse_ir_text(&context, cross_cut_program(), "promotion").expect("fixture parses"); + let contained = super::split_moved_functions(&module, &["wide".to_string()]) + .expect("the fixture splits") + .expect("`main` and `helper` stay behind"); + let sibling_ir = module.print_to_string().to_string(); + let contained_ir = contained.print_to_string().to_string(); + for name in ["counter", ".msg", "table"] { + let line = sibling_ir + .lines() + .find(|l| l.starts_with(&format!("@{name}"))) + .unwrap_or_else(|| panic!("`{name}` missing:\n{sibling_ir}")); + let is_promoted = line.starts_with(&format!("@{name}{PROMOTED_INFIX}")); + // `table` is referenced only from the sibling side: it stays local. + assert_eq!(is_promoted, name != "table", "{line}"); + if is_promoted { + assert!(line.contains(" hidden "), "promoted must be hidden: {line}"); + } + } + let helper = sibling_ir + .lines() + .find(|l| l.starts_with("define") && l.contains(&format!("@helper{PROMOTED_INFIX}"))) + .unwrap_or_else(|| panic!("`helper` not promoted:\n{sibling_ir}")); + assert!(helper.contains(" hidden "), "{helper}"); + assert!( + sibling_ir.contains("declare hidden i64 @wide"), + "the sibling side declares the moved function:\n{sibling_ir}" + ); + assert!( + contained_ir.contains("define hidden i64 @wide"), + "the contained side defines it:\n{contained_ir}" + ); + assert!( + !contained_ir.contains("define i32 @main") + && !contained_ir.contains("@table") + && !contained_ir.contains("@main"), + "the contained side holds nothing it does not reference:\n{contained_ir}" + ); + assert!( + contained_ir.contains("@counter") && contained_ir.contains("external hidden global i64"), + "{contained_ir}" + ); + + // A different unit (a different function set) gets different names. + let other_ir = cross_cut_program().replace("@main(", "@main2("); + let other = parse_ir_text(&context, &other_ir, "promotion_other").expect("parses"); + let _ = super::split_moved_functions(&other, &["wide".to_string()]) + .expect("splits") + .expect("has siblings"); + let token = |ir: &str| { + let at = ir.find(PROMOTED_INFIX).expect("a promoted name") + PROMOTED_INFIX.len(); + ir[at..at + 16].to_string() + }; + assert_ne!( + token(&sibling_ir), + token(&other.print_to_string().to_string()), + "two units must not share promoted names" + ); +} + +/// The sibling half keeps its statepoint stack map through the cut: the +/// collector must still find the roots of every function that stayed on the +/// optimized pipeline. (A contained function carries no statepoints of its +/// own: an over-budget statepoint function is re-lowered onto a shadow frame +/// first — see the next test.) +#[test] +fn the_sibling_half_keeps_its_stack_map() { + if !split_host() { + return; + } + let ir = r#" +declare i64 @may_collect() +declare i64 @leaf(i64) + +define i64 @narrow(i64 %a) gc "statepoint-example" { +entry: + %p = inttoptr i64 %a to ptr addrspace(1) + %t = call i64 @may_collect() + %bits = ptrtoint ptr addrspace(1) %p to i64 + %r = add i64 %t, %bits + ret i64 %r +} + +define i64 @wide(i64 %a) { +entry: + %t1 = call i64 @leaf(i64 %a) + %t2 = call i64 @leaf(i64 %t1) + %t3 = call i64 @leaf(i64 %t2) + %t4 = call i64 @leaf(i64 %t3) + %t5 = call i64 @leaf(i64 %t4) + %t6 = call i64 @leaf(i64 %t5) + %t7 = call i64 @leaf(i64 %t6) + %t8 = call i64 @leaf(i64 %t7) + %s1 = add i64 %t1, %t2 + %s2 = add i64 %s1, %t3 + %s3 = add i64 %s2, %t4 + %s4 = add i64 %s3, %t5 + %s5 = add i64 %s4, %t6 + %s6 = add i64 %s5, %t7 + %r = add i64 %s6, %t8 + ret i64 %r +} +"#; + global_init(&[]); + let target = host_target(); + let context = Context::create(); + let module = parse_ir_text(&context, ir, "gc_split").expect("fixture parses"); + let mut stats = UnitCodegenStats::default(); + let parts = with_test_fast_emit_budget_value(FastEmitBudget::Cap(12), || { + optimize_and_emit_module_with_stats( + &module, + &target, + &["-Os".into(), "-S".into()], + true, + Some(&mut stats), + ) + }) + .expect("emits"); + let over: Vec<&str> = stats + .fast_emit_fallbacks + .iter() + .map(|f| f.name.as_str()) + .collect(); + assert_eq!(over, ["wide"], "`wide` alone is over 12"); + assert_eq!(parts.len(), 2); + let names: Vec = + crate::gc_map::decode_stack_map_roots(std::str::from_utf8(&parts[0]).unwrap(), &target) + .expect("the sibling half's stack map decodes") + .into_iter() + .map(|(name, _)| name.trim_start_matches('_').to_string()) + .collect(); + assert_eq!(names, ["narrow"]); + assert!(!std::str::from_utf8(&parts[1]) + .unwrap() + .contains("llvm_stackmaps")); +} + +/// Over the budget, a statepoint function is not contained as it is: the +/// backend asks codegen to re-lower it onto a shadow frame (the typed retry +/// the RS4GC budgets use), so it can keep the optimized machine pipeline. +#[test] +fn an_over_budget_statepoint_function_requests_a_shadow_frame_relowering() { + let ir = r#" +declare i64 @may_collect() + +define i64 @wide(i64 %a) gc "statepoint-example" { +entry: + %p = inttoptr i64 %a to ptr addrspace(1) + %t1 = call i64 @may_collect() + %t2 = call i64 @may_collect() + %bits = ptrtoint ptr addrspace(1) %p to i64 + %s = add i64 %t1, %t2 + %r = add i64 %s, %bits + ret i64 %r +} +"#; + global_init(&[]); + let target = host_target(); + let context = Context::create(); + let module = parse_ir_text(&context, ir, "relower").expect("fixture parses"); + let err = with_test_fast_emit_budget_value(FastEmitBudget::Cap(1), || { + optimize_and_emit_module(&module, &target, &["-Os".into(), "-S".into()], true) + }) + .expect_err("an over-budget statepoint function must request a re-lowering"); + let retry = rs4gc_budget_retry(&err).expect("the request is typed"); + assert_eq!(retry.len(), 1); + assert_eq!(retry[0].name, "wide"); + assert!( + matches!(retry[0].cause, Rs4gcBudgetCause::MachineBudget { instructions } if instructions > 1), + "{:?}", + retry[0].cause + ); +} + +/// The bounded machine is chosen by how far past the budget the function +/// is: FastISel on the optimized pipeline up to four times over (inclusive), +/// O0 only beyond. The shipped path reports the tier it used. +#[test] +fn the_bounded_machine_grows_with_the_excess() { + assert_eq!(MachineTier::for_function(601, 600), MachineTier::FastIsel); + assert_eq!(MachineTier::for_function(2400, 600), MachineTier::FastIsel); + assert_eq!(MachineTier::for_function(2401, 600), MachineTier::O0); + + let with_wide = sibling_cost_fixture(true); + let tiers = |budget| { + global_init(&[]); + let target = host_target(); + let context = Context::create(); + let module = parse_ir_text(&context, &with_wide, "tiers").expect("fixture parses"); + let mut stats = UnitCodegenStats::default(); + let parts = with_test_fast_emit_budget_value(budget, || { + optimize_and_emit_module_with_stats( + &module, + &target, + &["-Os".into(), "-S".into()], + false, + Some(&mut stats), + ) + }) + .expect("emits"); + let asm = String::from_utf8(parts.last().unwrap().clone()).unwrap(); + (stats.fast_emit_fallbacks, function_assembly(&asm, "wide")) + }; + let (near, near_asm) = tiers(FastEmitBudget::Cap(7)); + assert_eq!(near.len(), 1); + assert_eq!(near[0].tier, MachineTier::FastIsel, "{near:?}"); + let (far, far_asm) = tiers(FastEmitBudget::Cap(1)); + let wide = far.iter().find(|f| f.name == "wide").expect("wide is over"); + assert_eq!(wide.tier, MachineTier::O0, "{far:?}"); + assert_ne!( + near_asm, far_asm, + "the two tiers must emit different machine code, or the tier is not live" + ); + // Code size is a property of large functions (the tier table on + // `MachineTier`); a fixture this small cannot show it, so only liveness + // of both tiers is asserted here. +} diff --git a/crates/perry-codegen/src/inprocess/optimize_emit.rs b/crates/perry-codegen/src/inprocess/optimize_emit.rs index edbe37ca32..3932b416fb 100644 --- a/crates/perry-codegen/src/inprocess/optimize_emit.rs +++ b/crates/perry-codegen/src/inprocess/optimize_emit.rs @@ -14,7 +14,7 @@ pub(super) fn optimize_and_emit( emit_asm: bool, native_roots: bool, mut stats: Option<&mut UnitCodegenStats>, -) -> Result> { +) -> Result>> { global_init(mllvm); announce(); @@ -81,13 +81,17 @@ pub(super) fn optimize_and_emit( // backend that can root an `invoke`, and since #7302 every call inside a // `try` is one — 26% of the gap suite (128 of 479 files) contains a `try`, // which the explicit bridge refuses outright (#7327/#7330). + // Functions RS4GC rewrites, and their pre-rewrite sizes: the machine + // budget below re-lowers an over-budget one of these onto a shadow frame. + let mut rewritten_functions = std::collections::HashSet::new(); + let mut pre_sizes = std::collections::HashMap::new(); if native_roots { // Sizes before the rewrite: the budget message below names them, and // the per-unit report compares them with the post-rewrite census. let budget = rs4gc_instruction_budget(); let preflight_cap = crate::codegen::helpers::root_spill_relocation_threshold(); - let rewritten_functions = rs4gc_functions(module); - let pre_sizes = if budget == RewriteBudget::Off && preflight_cap == 0 && stats.is_none() { + rewritten_functions = rs4gc_functions(module); + pre_sizes = if budget == RewriteBudget::Off && preflight_cap == 0 && stats.is_none() { std::collections::HashMap::new() } else { pre_rewrite_sizes(module) @@ -184,49 +188,140 @@ pub(super) fn optimize_and_emit( // The IR pipeline above has already done the requested optimization. For // an extreme generated function, LLVM's optimized *machine* pipeline can // still become super-linear in instruction selection / LiveIntervals / - // register allocation. Use an O0 target machine only for final emission - // of that unit; ordinary units keep `tm`, and the optimized IR is not - // rebuilt or demoted. + // register allocation, so such a function is emitted by a bounded target + // machine instead. The optimized IR is not rebuilt or demoted. // - // The selection is per function, the emission cannot be — a TargetMachine - // carries one optimization level for the whole module, and `optnone` does - // not reach LiveIntervals or the register allocator. Every ordinary - // function in the unit is demoted with the offender, which is why the - // budget sits above the measured population of extreme functions and why - // the log below names each of them. - let fast_emit = if opt == '0' { + // A TargetMachine carries one optimization level for a whole module, and + // `optnone` does not reach LiveIntervals or the register allocator, so the + // over-budget functions are first moved into a module of their own + // (`fast_emit_split`): the unit's ordinary functions keep `tm`, and only + // the moved ones pay for the bounded pipeline. Where the cut cannot be + // made (see `split_emission_supported` and `SplitDeclined`), the whole + // unit takes the bounded pipeline as it did before containment. + let mut fast_emit = if opt == '0' { Vec::new() } else { fast_emit_fallbacks(module, fast_emit_budget(effective_target)) }; + // First tier past the budget: a statepoint function is sent back to + // codegen and re-lowered with its GC roots in a shadow frame, then the + // unit is compiled again at the same optimization level through the + // optimized machine pipeline. Its size is mostly RS4GC's relocation + // fan-out, which the shadow frame does not have. Only a function that is + // still over the budget without statepoints (or never had them) reaches + // the bounded pipeline below. + let relower: Vec = fast_emit + .iter() + .filter(|f| rewritten_functions.contains(&f.name)) + .map(|f| Rs4gcBudgetViolation { + name: f.name.clone(), + pre_instructions: pre_sizes.get(&f.name).copied(), + cause: Rs4gcBudgetCause::MachineBudget { + instructions: f.instructions, + }, + cap: f.cap, + }) + .collect(); + if !relower.is_empty() { + crate::machine_tiers::note_relowered(relower.len()); + if let Ok(dir) = std::env::var("PERRY_LL_FAST_EMIT_DUMP") { + // The statepoint form of the unit, as the machine pipeline would + // have received it, for `llc` study of what the re-lowering saved. + dump_module(module, &dir, &relower[0].name, "relowered-unit"); + } + return Err(anyhow::Error::new(Rs4gcBudgetExceeded { + violations: relower, + })); + } + // `split`: `None` — no split (nothing over budget, or the cut declined); + // `Some(None)` — every function of the unit is over budget, so the whole + // unit is exactly the contained set; `Some(Some(m))` — `m` holds the + // moved functions and `module` the rest. + let split = + if fast_emit.is_empty() || !fast_emit_split::split_emission_supported(effective_target) { + None + } else { + let names: Vec = fast_emit.iter().map(|f| f.name.clone()).collect(); + match fast_emit_split::split_moved_functions(module, &names) { + Ok(contained) => Some(contained), + Err(declined) => { + eprintln!( + "perry: fast-emit containment declined for the unit of `{}`: {declined}", + names[0] + ); + None + } + } + }; + let whole_unit_is_contained = matches!(split, Some(None)); + let contained = split.flatten(); + if contained.is_some() || whole_unit_is_contained { + for fallback in &mut fast_emit { + fallback.contained = true; + } + } + // The bounded machine: the requested optimization level with FastISel + // instruction selection for a function up to `FAST_ISEL_TIER_FACTOR` + // times the budget, LLVM's O0 machine pipeline only past that (see + // `MachineTier`). + let tier = fast_emit + .first() + .map(|widest| MachineTier::for_function(widest.instructions, widest.cap)); + if let Some(tier) = tier { + for fallback in &mut fast_emit { + fallback.tier = tier; + } + } for fallback in &fast_emit { eprintln!("perry: {fallback}"); } + if let (Some(contained), Ok(dir)) = (&contained, std::env::var("PERRY_LL_FAST_EMIT_DUMP")) { + dump_module(contained, &dir, &fast_emit[0].name, "contained"); + } if let Some(stats) = stats.as_deref_mut() { stats.fast_emit_fallbacks = fast_emit.clone(); } - let fast_tm = if !fast_emit.is_empty() { - Some( - target + let fast_tm = match tier { + None => None, + Some(tier) => { + let level = match tier { + MachineTier::FastIsel => opt_level, + MachineTier::O0 => OptimizationLevel::None, + }; + let machine = target .create_target_machine( &triple, &cpu, &features, - OptimizationLevel::None, + level, RelocMode::PIC, CodeModel::Default, ) .ok_or_else(|| { anyhow!( - "failed to create bounded O0 emission TargetMachine for \ - `{effective_target}`" + "failed to create bounded emission TargetMachine for `{effective_target}`" ) - })?, - ) - } else { - None + })?; + if tier == MachineTier::FastIsel { + // SAFETY: `machine` is a live TargetMachine owned by this frame. + unsafe { + llvm_sys::target_machine::LLVMSetTargetMachineFastISel(machine.as_mut_ptr(), 1) + }; + } + if contained.is_some() || whole_unit_is_contained { + crate::machine_tiers::note_contained(tier, fast_emit.len()); + } else { + crate::machine_tiers::note_whole_unit(tier, fast_emit[0].unit_functions); + } + Some(machine) + } + }; + // Contained: the unit keeps `tm` and only the moved functions use the + // bounded machine. Not contained: the whole unit uses it. + let unit_tm = match (&contained, &fast_tm) { + (None, Some(fast_tm)) => fast_tm, + _ => &tm, }; - let emit_tm = fast_tm.as_ref().unwrap_or(&tm); let kind = if emit_asm { FileType::Assembly @@ -234,13 +329,51 @@ pub(super) fn optimize_and_emit( FileType::Object }; let emit_started = std::time::Instant::now(); - let obj = emit_tm + let mut parts = Vec::with_capacity(2); + let obj = unit_tm .write_to_memory_buffer(module, kind) .map_err(|e| anyhow!("{kind:?} emission failed:\n{}", e.to_string()))?; + parts.push(obj.as_slice().to_vec()); + drop(obj); + if let (Some(contained), Some(fast_tm)) = (&contained, &fast_tm) { + let obj = fast_tm + .write_to_memory_buffer(contained, kind) + .map_err(|e| { + anyhow!( + "{kind:?} emission of the contained over-budget functions failed:\n{}", + e.to_string() + ) + })?; + parts.push(obj.as_slice().to_vec()); + } if let Some(stats) = stats { stats.emit_secs = emit_started.elapsed().as_secs_f64(); } - Ok(obj.as_slice().to_vec()) + Ok(parts) +} + +/// `PERRY_LL_FAST_EMIT_DUMP=`: keep the bitcode of a contained module +/// (or of a unit whose statepoint functions are being re-lowered), so the +/// machine pipeline of an extreme function can be studied with `llc` +/// without recompiling the program that produced it. Diagnostic only; it +/// does not change what is emitted. +fn dump_module(module: &inkwell::module::Module<'_>, dir: &str, first: &str, kind: &str) { + let safe: String = first + .chars() + .map(|c| { + if c.is_ascii_alphanumeric() || c == '_' { + c + } else { + '_' + } + }) + .take(120) + .collect(); + let path = std::path::Path::new(dir).join(format!("{safe}.{kind}.bc")); + let _ = std::fs::create_dir_all(dir); + if module.write_bitcode_to_path(&path) { + eprintln!("perry: kept {kind} fast-emit module: {}", path.display()); + } } #[cfg(test)] @@ -1188,9 +1321,11 @@ entry: name: "wide".to_string(), instructions: 9, cap: 8, - // `narrow` is under the cap and `sink` is a declaration; both - // are still emitted by the demoted machine pipeline. + // `narrow` is under the cap and `sink` is a declaration. unit_functions: 2, + contained: false, + // 9 is within four times 8. + tier: MachineTier::FastIsel, }], "only the function over the budget is selected" ); @@ -1210,7 +1345,7 @@ entry: "9 instructions", "budget 8", "requested IR optimization", - "O0 machine pipeline", + "optimized machine pipeline with FastISel", "all 2 of its defined functions", "PERRY_LL_FAST_EMIT_MAX_INSTRS", ] { @@ -1423,131 +1558,4 @@ entry: .expect("-O0 emits"); assert!(stats.fast_emit_fallbacks.is_empty()); } - - /// One function over the budget, one ordinary function that is nowhere - /// near it, and one external callee so nothing folds away. `narrow` - /// holds three values across three calls, which is what makes its - /// machine code differ between the optimized and the O0 register - /// allocators. - fn sibling_cost_fixture(with_wide: bool) -> String { - let wide = r#" -define i64 @wide(i64 %n) { -entry: - %a = call i64 @src(i64 %n) - %b = call i64 @src(i64 %a) - %c = call i64 @src(i64 %b) - %d = call i64 @src(i64 %c) - %s = add i64 %a, %b - %t = add i64 %s, %c - %u = add i64 %t, %d - ret i64 %u -} -"#; - format!( - r#" -declare i64 @src(i64) - -define i64 @narrow(i64 %x, i64 %y) {{ -entry: - %a = call i64 @src(i64 %x) - %b = call i64 @src(i64 %y) - %c = call i64 @src(i64 %a) - %s = add i64 %a, %b - %t = add i64 %s, %c - ret i64 %t -}} -{}"#, - if with_wide { wide } else { "" } - ) - } - - /// The assembly of one function, from its label to the end of its body. - /// Tolerates ELF (`narrow:` / `.size`) and Mach-O (`_narrow:`) spelling. - fn function_assembly(asm: &str, name: &str) -> String { - let label_elf = format!("{name}:"); - let label_macho = format!("_{name}:"); - let mut body: Vec<&str> = Vec::new(); - let mut inside = false; - for line in asm.lines() { - let trimmed = line.trim(); - if !inside { - inside = trimmed == label_elf || trimmed == label_macho; - continue; - } - let next_symbol = trimmed.ends_with(':') - && !trimmed.starts_with('.') - && !trimmed.starts_with('L') - && !trimmed.contains(' '); - if trimmed.starts_with(".size") || trimmed == ".cfi_endproc" || next_symbol { - break; - } - body.push(line); - } - assert!( - body.len() > 3, - "no body extracted for `{name}` — the assertion below would be vacuous:\n{asm}" - ); - body.join("\n") - } - - fn emit_assembly(ir: &str, module_name: &str, budget: FastEmitBudget) -> (String, Vec) { - global_init(&[]); - let target = crate::codegen::default_target_triple(); - let context = Context::create(); - let module = parse_ir_text(&context, ir, module_name).expect("fixture parses"); - let mut stats = UnitCodegenStats::default(); - let asm = with_test_fast_emit_budget_value(budget, || { - optimize_and_emit_module_with_stats( - &module, - &target, - &["-Os".into(), "-S".into()], - false, - Some(&mut stats), - ) - }) - .expect("the fixture emits"); - ( - String::from_utf8(asm).expect("LLVM emits UTF-8 assembly"), - stats - .fast_emit_fallbacks - .iter() - .map(|f| f.name.clone()) - .collect(), - ) - } - - /// What the budget actually costs, and why it is calibrated above the - /// measured population instead of below the smallest pathological case: - /// the demotion is a whole-unit act. An ordinary function emits the same - /// machine code whether or not an extreme function shares its unit — but - /// only while the unit keeps the optimized machine pipeline. Cross the - /// budget and that ordinary function's code changes too, without ever - /// having been over any budget itself. - #[test] - fn the_budget_is_what_makes_ordinary_siblings_pay() { - let with_wide = sibling_cost_fixture(true); - let alone = sibling_cost_fixture(false); - - let (undemoted, none) = emit_assembly(&with_wide, "sibling_cost_ok", FastEmitBudget::Off); - assert!(none.is_empty(), "this arm must not demote: {none:?}"); - let (solo, _) = emit_assembly(&alone, "sibling_cost_alone", FastEmitBudget::Off); - assert_eq!( - function_assembly(&undemoted, "narrow"), - function_assembly(&solo, "narrow"), - "an undemoted unit emits an ordinary function exactly as a unit without the \ - extreme function does" - ); - - // Same module, same IR pipeline, budget crossed: `narrow` is not over - // it and is compiled differently anyway. - let (demoted, over) = - emit_assembly(&with_wide, "sibling_cost_demoted", FastEmitBudget::Cap(7)); - assert_eq!(over, ["wide"], "only `wide` is over the budget"); - assert_ne!( - function_assembly(&demoted, "narrow"), - function_assembly(&undemoted, "narrow"), - "if the demotion did not reach `narrow`, LLVM grew a per-function escape from the \ - optimized machine pipeline and this budget can move back down" - ); - } } diff --git a/crates/perry-codegen/src/lib.rs b/crates/perry-codegen/src/lib.rs index d687576831..f234c7f53c 100644 --- a/crates/perry-codegen/src/lib.rs +++ b/crates/perry-codegen/src/lib.rs @@ -28,6 +28,7 @@ pub(crate) mod lower_call; pub(crate) mod lower_conditional; pub(crate) mod lower_string_concat; pub(crate) mod lower_string_method; +pub mod machine_tiers; /// #463/#512 dispatch-table/manifest drift check — see the module docs /// for why this moved here from an integration test (#10668's 15-row /// drift, which nothing caught until it surfaced on an unrelated PR). diff --git a/crates/perry-codegen/src/linker.rs b/crates/perry-codegen/src/linker.rs index 53b0a53eab..f1e1d2f1ba 100644 --- a/crates/perry-codegen/src/linker.rs +++ b/crates/perry-codegen/src/linker.rs @@ -691,8 +691,33 @@ pub(crate) fn native_plan_args( /// /// A plan without `-S` (the non-statepoint backends) passes its bytes through /// untouched. +/// +/// `parts` is what `optimize_and_emit_module` returned: one emission, or two +/// when fast-emit containment moved over-budget functions into a module of +/// their own. Each part is finished on its own (each carries its own stack +/// map) and two parts are joined the way codegen units are. #[cfg(feature = "llvm-inprocess")] pub(crate) fn finish_native_emission( + parts: Vec>, + effective_target: &str, + clang_args: &[String], +) -> Result> { + let mut objects = Vec::with_capacity(parts.len()); + for part in parts { + objects.push(finish_native_emission_part( + part, + effective_target, + clang_args, + )?); + } + if objects.len() == 1 { + return Ok(objects.pop().expect("one part")); + } + merge_unit_objects(&objects) +} + +#[cfg(feature = "llvm-inprocess")] +fn finish_native_emission_part( bytes: Vec, effective_target: &str, clang_args: &[String], @@ -812,13 +837,29 @@ fn compile_ll_inprocess_in( metadata_path.display() ); } - match crate::inprocess::compile_ll_to_object_inprocess( + let emitted = crate::inprocess::compile_ll_to_object_inprocess( ll_text, &plan.effective_target, &plan.clang_args, &module_name, native_roots, - ) { + ); + // Fast-emit containment split the module (see `inprocess::fast_emit_split`): + // finish and join the parts like codegen units. The single-part arms below + // keep their scratch-file and PERRY_LLVM_KEEP_IR behaviour. + let emitted = match emitted { + Ok(parts) if parts.len() > 1 => { + let object = finish_native_emission(parts, &plan.effective_target, &plan.clang_args) + .map_err(|error| failed_scratch.finish_with_ir(error, ll_text))?; + if !policy.keep { + let _ = fs::remove_dir_all(&paths.scratch_dir); + } + return Ok(object); + } + Ok(mut parts) => Ok(parts.pop().expect("an emission has at least one part")), + Err(error) => Err(error), + }; + match emitted { // Statepoint plans ask for `-S`: #7314's compact-map rewriter operates // on assembly. Rewrite and assemble those bytes before returning them. Ok(bytes) if plan.asm_path.is_some() => { diff --git a/crates/perry-codegen/src/machine_tiers.rs b/crates/perry-codegen/src/machine_tiers.rs new file mode 100644 index 0000000000..9674e1c3b6 --- /dev/null +++ b/crates/perry-codegen/src/machine_tiers.rs @@ -0,0 +1,136 @@ +//! Which machine-code tier each function of a compile landed in. +//! +//! A function over the optimized machine-pipeline budget +//! (`inprocess::DEFAULT_FAST_EMIT_MAX_INSTRS_X86_64`) no longer silently +//! costs its whole codegen unit. It takes, in order: +//! +//! 1. **re-lowered** — a statepoint function is re-lowered with its GC roots +//! in a shadow frame and compiled again, still through the optimized +//! machine pipeline; +//! 2. **contained** — a function still over the budget is emitted alone +//! through a bounded machine (`inprocess::MachineTier`: the optimized +//! pipeline with FastISel, or O0 past four times the budget), and its +//! unit's other functions keep the optimized one; +//! 3. **whole unit** — where the unit cannot be split (COFF, or a host that +//! cannot partially link the target's objects), the whole unit takes the +//! bounded machine. +//! +//! These counters are process-wide and only ever grow. The compile driver +//! prints [`summary`] once codegen is done, so a build in which anything +//! left the optimized tier says so in one line instead of burying it in +//! per-unit logs. + +use std::sync::atomic::{AtomicUsize, Ordering}; + +static RELOWERED: AtomicUsize = AtomicUsize::new(0); +static CONTAINED_FAST_ISEL: AtomicUsize = AtomicUsize::new(0); +static CONTAINED_O0: AtomicUsize = AtomicUsize::new(0); +static WHOLE_UNITS: AtomicUsize = AtomicUsize::new(0); +static WHOLE_UNIT_FUNCTIONS: AtomicUsize = AtomicUsize::new(0); +static WHOLE_UNITS_O0: AtomicUsize = AtomicUsize::new(0); + +/// A snapshot of the counters. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct MachineTierCounts { + /// Statepoint functions re-lowered onto a shadow frame because they were + /// over the machine budget after IR optimization. + pub relowered: usize, + /// Functions emitted alone through the optimized machine pipeline with + /// FastISel instruction selection. + pub contained_fast_isel: usize, + /// Functions emitted alone through LLVM's O0 machine pipeline (more than + /// four times over the budget). + pub contained_o0: usize, + /// Units emitted whole through a bounded machine because they could not + /// be split. + pub whole_units: usize, + /// Of those, the units that took the O0 machine pipeline. + pub whole_units_o0: usize, + /// Defined functions in those units. + pub whole_unit_functions: usize, +} + +pub(crate) fn note_relowered(n: usize) { + RELOWERED.fetch_add(n, Ordering::Relaxed); +} + +#[cfg(feature = "llvm-inprocess")] +pub(crate) fn note_contained(tier: crate::inprocess::MachineTier, n: usize) { + match tier { + crate::inprocess::MachineTier::FastIsel => &CONTAINED_FAST_ISEL, + crate::inprocess::MachineTier::O0 => &CONTAINED_O0, + } + .fetch_add(n, Ordering::Relaxed); +} + +#[cfg(feature = "llvm-inprocess")] +pub(crate) fn note_whole_unit(tier: crate::inprocess::MachineTier, functions: usize) { + WHOLE_UNITS.fetch_add(1, Ordering::Relaxed); + WHOLE_UNIT_FUNCTIONS.fetch_add(functions, Ordering::Relaxed); + if tier == crate::inprocess::MachineTier::O0 { + WHOLE_UNITS_O0.fetch_add(1, Ordering::Relaxed); + } +} + +pub fn counts() -> MachineTierCounts { + MachineTierCounts { + relowered: RELOWERED.load(Ordering::Relaxed), + contained_fast_isel: CONTAINED_FAST_ISEL.load(Ordering::Relaxed), + contained_o0: CONTAINED_O0.load(Ordering::Relaxed), + whole_units: WHOLE_UNITS.load(Ordering::Relaxed), + whole_units_o0: WHOLE_UNITS_O0.load(Ordering::Relaxed), + whole_unit_functions: WHOLE_UNIT_FUNCTIONS.load(Ordering::Relaxed), + } +} + +/// One line for the compile summary, or `None` when every function this +/// process emitted kept the optimized machine pipeline. +pub fn summary() -> Option { + format_summary(counts()) +} + +fn format_summary(c: MachineTierCounts) -> Option { + if c == MachineTierCounts::default() { + return None; + } + Some(format!( + "machine code: {} function(s) over the optimized-pipeline budget re-lowered onto a \ + shadow frame (optimized pipeline kept); {} emitted alone with FastISel, {} alone \ + with O0; {} unit(s) / {} function(s) emitted whole by a bounded machine ({} of those \ + units with O0) because the unit could not be split", + c.relowered, + c.contained_fast_isel, + c.contained_o0, + c.whole_units, + c.whole_unit_functions, + c.whole_units_o0 + )) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn summary_is_silent_until_a_function_leaves_the_optimized_tier() { + assert_eq!(format_summary(MachineTierCounts::default()), None); + let line = format_summary(MachineTierCounts { + relowered: 1, + contained_fast_isel: 2, + contained_o0: 4, + whole_units: 3, + whole_unit_functions: 950, + whole_units_o0: 1, + }) + .expect("a non-empty census prints"); + for needle in [ + "1 function(s)", + "2 emitted alone with FastISel", + "4 alone with O0", + "3 unit(s) / 950 function(s)", + "1 of those units with O0", + ] { + assert!(line.contains(needle), "{needle:?} missing from {line}"); + } + } +} diff --git a/crates/perry-codegen/src/native_emit.rs b/crates/perry-codegen/src/native_emit.rs index c8ce7a3315..1dd2ef406b 100644 --- a/crates/perry-codegen/src/native_emit.rs +++ b/crates/perry-codegen/src/native_emit.rs @@ -238,6 +238,14 @@ pub(crate) fn apply_budget_spill_retry<'a>( estimated_relocations, violation.cap, ), + crate::inprocess::Rs4gcBudgetCause::MachineBudget { instructions } => { + eprintln!( + "perry: `{}` has {} instructions after IR optimization, above the \ + optimized machine-pipeline budget {}; retrying it with precise GC roots \ + in a shadow frame so it keeps the optimized machine pipeline", + violation.name, instructions, violation.cap, + ); + } crate::inprocess::Rs4gcBudgetCause::PostRewrite { post_instructions } => { eprintln!( "perry: `{}` exceeded the post-RS4GC instruction budget ({} -> {} \ @@ -1388,6 +1396,34 @@ mod tests { assert!(after.contains("@js_shadow_frame_pop"), "{after}"); } + /// The machine-pipeline budget's first tier: a statepoint function over + /// it after IR optimization is re-lowered onto a shadow frame and the + /// unit is compiled again, instead of being sent to the bounded (O0) + /// machine pipeline with its relocations. A one-instruction cap keeps + /// the fixture small; the shadow-frame IR proves the retry happened. + #[test] + fn machine_budget_relowers_a_statepoint_function_onto_a_shadow_frame() { + let _native = crate::codegen::helpers::NativeRootsPin::native(); + let mut module = precise_root_fixture(false); + let object = crate::inprocess::with_test_fast_emit_budget(1, || { + compile_module_native(&mut module, None, "machine_budget_retry_fixture") + }) + .expect("a machine-budget miss must re-lower and finish emission"); + assert!(!object.is_empty()); + let retried = module + .deduped_function_refs() + .into_iter() + .find(|function| function.name == "native_root_diff_fixture") + .expect("fixture function survives the retry"); + assert!( + retried.spills_roots_to_shadow_frame(), + "the over-budget statepoint function must be re-lowered, not demoted as it is" + ); + let after = retried.to_ir(); + assert!(!after.contains("gc \"statepoint-example\""), "{after}"); + assert!(after.contains("@js_shadow_frame_enter"), "{after}"); + } + /// The reported Claude bundle takes the split-unit worker path. Its retry /// source must stay on the producer thread (the `LlFunction` graph is not /// `Send`) while LLVM reports the typed violation from a worker. A compact @@ -1619,25 +1655,28 @@ fn compile_module_diff_once( &args, native_roots, )?; + // One part normally, two under fast-emit containment; the verdict + // and the dump are over their concatenation. + let flat = |parts: &[Vec]| parts.concat(); if bytes_text == bytes_native { eprintln!( "perry: [ir-diff] OK — native and text arms emit byte-identical objects \ ({} bytes)", - bytes_text.len() + flat(&bytes_text).len() ); } else { eprintln!( "perry: [ir-diff] MISMATCH — object bytes differ (text {} vs native {}); \ set PERRY_LLVM_DIFF_DIR to dump both arms' pre-opt IR", - bytes_text.len(), - bytes_native.len() + flat(&bytes_text).len(), + flat(&bytes_native).len() ); if let Some(dir) = &dump_dir { let _ = std::fs::create_dir_all(dir); let _ = std::fs::write(format!("{dir}/text_arm.ll"), &pre_text); let _ = std::fs::write(format!("{dir}/native_arm.ll"), &pre_native); - let _ = std::fs::write(format!("{dir}/text_arm.o"), &bytes_text); - let _ = std::fs::write(format!("{dir}/native_arm.o"), &bytes_native); + let _ = std::fs::write(format!("{dir}/text_arm.o"), flat(&bytes_text)); + let _ = std::fs::write(format!("{dir}/native_arm.o"), flat(&bytes_native)); eprintln!("perry: [ir-diff] arms dumped under {dir}"); } } diff --git a/crates/perry-codegen/src/native_root_coverage/mod.rs b/crates/perry-codegen/src/native_root_coverage/mod.rs index c260bfd69f..9be0645078 100644 --- a/crates/perry-codegen/src/native_root_coverage/mod.rs +++ b/crates/perry-codegen/src/native_root_coverage/mod.rs @@ -543,7 +543,8 @@ pub(crate) fn assembly_for(ir: &str, target: &str) -> String { true, ) .unwrap_or_else(|e| panic!("assembly emission failed for {target}: {e:#}")); - String::from_utf8(bytes).expect("assembler text should be UTF-8") + assert_eq!(bytes.len(), 1, "an -O0 emission is never split"); + String::from_utf8(bytes.concat()).expect("assembler text should be UTF-8") } /// Per-safepoint root lists for `symbol` from the compact map the binary ships. diff --git a/crates/perry/src/commands/compile/build_cache.rs b/crates/perry/src/commands/compile/build_cache.rs index 825a1e4252..2aa5a7b7d4 100644 --- a/crates/perry/src/commands/compile/build_cache.rs +++ b/crates/perry/src/commands/compile/build_cache.rs @@ -248,6 +248,9 @@ const BUILD_CACHE_ENV_EXCLUSIONS: &[&str] = &[ // Human-facing telemetry only; never changes IR or object bytes. "PERRY_CODEGEN_PROGRESS", "PERRY_CODEGEN_UNIT_TIMINGS", + // Writes the contained over-budget module's bitcode next to the build for + // `llc` study; the emitted objects are the same with it on and off. + "PERRY_LL_FAST_EMIT_DUMP", // `packed_loop_reject` prints the admission chain's declining condition and // returns `None` either way — the rejection is what the caller already got // without the flag, so the emitted code is identical. An input, rather than diff --git a/crates/perry/src/commands/compile/run_pipeline.rs b/crates/perry/src/commands/compile/run_pipeline.rs index 79b4463a89..6ad1398617 100644 --- a/crates/perry/src/commands/compile/run_pipeline.rs +++ b/crates/perry/src/commands/compile/run_pipeline.rs @@ -6029,6 +6029,12 @@ pub fn run_with_parse_cache( } } + // Any function that left the optimized machine pipeline (see + // `perry_codegen::machine_tiers`), in one line. + if let Some(line) = perry_codegen::machine_tiers::summary() { + eprintln!("perry: {line}"); + } + // ── Loud failure summary ───────────────────────────────────────── // // Render the per-module compile errors prominently *here*, before