From 7ca5571cbe67a3bfef6f91b48d9e8b6764180bc3 Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Thu, 1 Oct 2026 19:55:38 +0300
Subject: [PATCH 01/11] Performance gate: per-PR baseline overlays, so
baselines stop conflicting
Every baseline sat in one vm/selfhost/perf-baseline.json. Any branch that met a
new runner CPU or moved a benchmark edited it, unrelated rows sat within git's
context of each other, and two branches calibrating the same CPU conflicted by
construction -- eleven open branches carried edits to it.
vm/selfhost/perf-baseline/ now holds policy.json, base/.json and
pr/.json. A pull request writes only its own overlay, so the change
stays in its diff and no other branch can touch the file. Calibrations of the
same CPU combine; a rebaseline records the value it replaces, so two changes
moving one benchmark are reported by name instead of as a JSON conflict. A
nightly fold moves merged overlays into base/ and refuses to commit unless the
resolved baseline is unchanged. The migration is lossless (asserted).
An improvement past tolerance now fails the gate until it is rebaselined, and
calibration tolerances learn the spread in both directions. The Port Status page
gains a ParparVM vs JDK 25 table rendered from these baselines at site-build
time; the absolute-time table stays for ports with no JDK arm.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
.github/workflows/linux-build-run.yml | 4 +
.github/workflows/parparvm-selfhost.yml | 2 +-
.github/workflows/parparvm-tests-windows.yml | 8 +
.github/workflows/perf-baseline.yml | 111 +
.github/workflows/scripts-macos.yml | 4 +
.gitignore | 3 +
.../assets/css/extended/cn1-port-status.css | 15 +
.../website/layouts/_default/port-status.html | 56 +
scripts/website/build.sh | 7 +
scripts/website/preview.sh | 3 +
scripts/website/validate_port_status.mjs | 20 +
vm/selfhost/README.md | 37 +-
vm/selfhost/calibrate-perf-baseline.py | 240 +-
vm/selfhost/ci-perf-gate.sh | 28 +-
vm/selfhost/perf-baseline.json | 1930 -----------------
vm/selfhost/perf-baseline/README.md | 11 +
.../base/linux-arm64@neoverse-n2.json | 102 +
...x-x64@amd-epyc-7763-64-core-processor.json | 136 ++
...x-x64@amd-epyc-9v45-96-core-processor.json | 136 ++
...x-x64@amd-epyc-9v74-80-core-processor.json | 136 ++
.../base/linux-x64@intel-xeon-6973p-c.json | 136 ++
.../linux-x64@intel-xeon-platinum-8370c.json | 136 ++
.../linux-x64@intel-xeon-platinum-8573c.json | 136 ++
.../perf-baseline/base/macos-arm64.json | 124 ++
...ily-8-model-d49-microsoft-corporation.json | 96 +
...ily-8-model-d84-microsoft-corporation.json | 136 ++
...@amd64-family-25-model-1-authenticamd.json | 100 +
...amd64-family-25-model-17-authenticamd.json | 136 ++
...@amd64-family-26-model-2-authenticamd.json | 136 ++
...tel64-family-6-model-106-genuineintel.json | 136 ++
...tel64-family-6-model-207-genuineintel.json | 136 ++
vm/selfhost/perf-baseline/policy.json | 10 +
vm/selfhost/perf-gate.py | 139 +-
vm/selfhost/perf_baseline.py | 581 +++++
vm/selfhost/test_perf_gate.py | 408 +++-
35 files changed, 3395 insertions(+), 2140 deletions(-)
create mode 100644 .github/workflows/perf-baseline.yml
delete mode 100644 vm/selfhost/perf-baseline.json
create mode 100644 vm/selfhost/perf-baseline/README.md
create mode 100644 vm/selfhost/perf-baseline/base/linux-arm64@neoverse-n2.json
create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-7763-64-core-processor.json
create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v45-96-core-processor.json
create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v74-80-core-processor.json
create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-6973p-c.json
create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8370c.json
create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8573c.json
create mode 100644 vm/selfhost/perf-baseline/base/macos-arm64.json
create mode 100644 vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d49-microsoft-corporation.json
create mode 100644 vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d84-microsoft-corporation.json
create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-1-authenticamd.json
create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-17-authenticamd.json
create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@amd64-family-26-model-2-authenticamd.json
create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-106-genuineintel.json
create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-207-genuineintel.json
create mode 100644 vm/selfhost/perf-baseline/policy.json
create mode 100644 vm/selfhost/perf_baseline.py
diff --git a/.github/workflows/linux-build-run.yml b/.github/workflows/linux-build-run.yml
index c1e86edfab0..27259c7e8c9 100644
--- a/.github/workflows/linux-build-run.yml
+++ b/.github/workflows/linux-build-run.yml
@@ -396,6 +396,10 @@ jobs:
# results travel with the screenshots.
- name: ParparVM vs JDK 25 (performance gate)
if: success() || failure()
+ # Names this pull request's overlay (perf-baseline/pr/.json) in the
+ # comment. A workflow_dispatch run carries no pull request, so the gate asks gh.
+ env:
+ GH_TOKEN: ${{ github.token }}
run: |
vm/selfhost/ci-perf-gate.sh run linux-${{ matrix.arch }} artifacts/linux-port/raw/perf \
--hello-workload "$GITHUB_WORKSPACE/artifacts/linux-port/raw/translation-record.jsonl" \
diff --git a/.github/workflows/parparvm-selfhost.yml b/.github/workflows/parparvm-selfhost.yml
index 6818d5577d0..322b8b0ee70 100644
--- a/.github/workflows/parparvm-selfhost.yml
+++ b/.github/workflows/parparvm-selfhost.yml
@@ -133,7 +133,7 @@ jobs:
# - Keeping it in both places would pay for the expensive -O3 self-host build
# twice on any PR carrying the `selfhost` label.
#
- # The baselines and tolerance live in vm/selfhost/perf-baseline.json, the
+ # The baselines and tolerance live in vm/selfhost/perf-baseline/, the
# measurement in vm/selfhost/perf-gate.py.
# Both trees, so a divergence can be inspected rather than guessed at from
diff --git a/.github/workflows/parparvm-tests-windows.yml b/.github/workflows/parparvm-tests-windows.yml
index 0bd6e65bc04..669fd8a6cdc 100644
--- a/.github/workflows/parparvm-tests-windows.yml
+++ b/.github/workflows/parparvm-tests-windows.yml
@@ -558,6 +558,10 @@ jobs:
# builds only ask javac for -source/-target 1.8 against an explicit bootclasspath.
- name: ParparVM vs JDK 25 (performance gate, x64)
if: success() || failure()
+ # Names this pull request's overlay (perf-baseline/pr/.json) in the
+ # comment. A workflow_dispatch run carries no pull request, so the gate asks gh.
+ env:
+ GH_TOKEN: ${{ github.token }}
shell: bash
run: |
export JDK_8_HOME="${JDK_8_HOME:-$JAVA_HOME}"
@@ -763,6 +767,10 @@ jobs:
# builds only ask javac for -source/-target 1.8 against an explicit bootclasspath.
- name: ParparVM vs JDK 25 (performance gate, arm64)
if: success() || failure()
+ # Names this pull request's overlay (perf-baseline/pr/.json) in the
+ # comment. A workflow_dispatch run carries no pull request, so the gate asks gh.
+ env:
+ GH_TOKEN: ${{ github.token }}
shell: bash
run: |
export JDK_8_HOME="${JDK_8_HOME:-$JAVA_HOME}"
diff --git a/.github/workflows/perf-baseline.yml b/.github/workflows/perf-baseline.yml
new file mode 100644
index 00000000000..5ad54b50169
--- /dev/null
+++ b/.github/workflows/perf-baseline.yml
@@ -0,0 +1,111 @@
+name: Performance baselines
+
+# The ParparVM performance gate's baselines live in vm/selfhost/perf-baseline/:
+# base/ holds the consolidated rows, and each pull request that changes a baseline
+# writes only its own pr/.json, so baseline changes stay in the pull
+# request's diff without two branches ever editing the same file
+# (vm/selfhost/perf_baseline.py explains the layout).
+#
+# check every pull request: the overlays resolve against base/ without
+# contradicting each other, and the pull request touched only its own
+# overlay (base/ is the fold's alone). No paths filter: several of this
+# repository's pull requests exceed GitHub's paths-filter diff limit, and a
+# filtered check would silently not run on exactly those.
+# fold nightly on master: every overlay on master belongs to a merged pull
+# request, so it is folded into base/ and deleted, in one commit that
+# provably changes no verdict. Then the website is rebuilt, because the
+# Port Status page's ParparVM vs JDK 25 table is rendered from these rows.
+# A push made with GITHUB_TOKEN triggers no workflow, hence the dispatch.
+
+on:
+ pull_request:
+ branches:
+ - master
+ push:
+ branches:
+ - master
+ schedule:
+ - cron: '20 3 * * *'
+ workflow_dispatch:
+
+permissions:
+ contents: read
+
+concurrency:
+ group: ${{ github.workflow }}-${{ github.event_name }}-${{ github.event.pull_request.number || github.ref }}
+ cancel-in-progress: ${{ github.event_name == 'pull_request' }}
+
+jobs:
+ check:
+ if: github.event_name == 'pull_request' || github.event_name == 'push'
+ runs-on: ubuntu-24.04
+ steps:
+ - uses: actions/checkout@v6
+ with:
+ # The pull request's base commit has to be present for the diff.
+ fetch-depth: 0
+
+ - name: The gate's own unit tests
+ working-directory: vm/selfhost
+ run: python3 -m unittest test_perf_gate
+
+ - name: Check the baselines
+ env:
+ BASE_SHA: ${{ github.event.pull_request.base.sha }}
+ PR_NUMBER: ${{ github.event.pull_request.number }}
+ run: |
+ if [ -n "${BASE_SHA}" ]; then
+ python3 vm/selfhost/perf_baseline.py check --base "${BASE_SHA}" --pr "${PR_NUMBER}"
+ else
+ python3 vm/selfhost/perf_baseline.py check
+ fi
+
+ fold:
+ if: (github.event_name == 'schedule' || github.event_name == 'workflow_dispatch') && github.ref == 'refs/heads/master'
+ runs-on: ubuntu-24.04
+ permissions:
+ contents: write
+ actions: write
+ concurrency:
+ group: perf-baseline-fold
+ cancel-in-progress: false
+ steps:
+ - uses: actions/checkout@v6
+ with:
+ ref: master
+
+ - name: Fold merged overlays into base/
+ id: fold
+ run: |
+ set -euo pipefail
+ git config user.name "github-actions[bot]"
+ git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
+ # A merge landing while this runs makes the push non-fast-forward. Start over
+ # from the new master rather than rebasing: the fold is cheap, and redoing it
+ # folds the overlay that just arrived as well.
+ for attempt in 1 2 3; do
+ git fetch --quiet origin master
+ git reset --quiet --hard origin/master
+ python3 vm/selfhost/perf_baseline.py fold
+ git add -A vm/selfhost/perf-baseline
+ if git diff --staged --quiet; then
+ echo "No overlays to fold."
+ echo "folded=false" >> "$GITHUB_OUTPUT"
+ exit 0
+ fi
+ git diff --staged --stat
+ git commit --quiet -m "ci: fold merged performance baselines into perf-baseline/base"
+ if git push origin HEAD:master; then
+ echo "folded=true" >> "$GITHUB_OUTPUT"
+ exit 0
+ fi
+ echo "master moved during the fold (attempt ${attempt}); retrying."
+ done
+ echo "Could not push the fold after 3 attempts." >&2
+ exit 1
+
+ - name: Rebuild the Port Status page
+ if: steps.fold.outputs.folded == 'true'
+ env:
+ GH_TOKEN: ${{ github.token }}
+ run: gh workflow run website-docs.yml --ref master -f deploy_production=true
diff --git a/.github/workflows/scripts-macos.yml b/.github/workflows/scripts-macos.yml
index efa6a5794c9..b0d3cb8f7f1 100644
--- a/.github/workflows/scripts-macos.yml
+++ b/.github/workflows/scripts-macos.yml
@@ -247,6 +247,10 @@ jobs:
# got it, and JDK 25 is fetched privately so no later step sees a different JDK.
- name: ParparVM vs JDK 25 (performance gate)
if: success() || failure()
+ # Names this pull request's overlay (perf-baseline/pr/.json) in the
+ # comment. A workflow_dispatch run carries no pull request, so the gate asks gh.
+ env:
+ GH_TOKEN: ${{ github.token }}
run: |
set -u
ENV_FILE="${TMPDIR%/}/codenameone-tools/tools/env.sh"
diff --git a/.gitignore b/.gitignore
index 50a08a7d25c..50246cad5ee 100644
--- a/.gitignore
+++ b/.gitignore
@@ -193,3 +193,6 @@ cn1-build-hints.json
# Concatenated CSS the native-theme build feeds the compiler. Regenerated on
# every run; the parts under native-themes//*.css are the source.
native-themes/*/target/
+
+# Rendered from vm/selfhost/perf-baseline by scripts/website/build.sh; never committed.
+docs/website/data/port_status_jdk25.json
diff --git a/docs/website/assets/css/extended/cn1-port-status.css b/docs/website/assets/css/extended/cn1-port-status.css
index 20e7179f398..8d48938ad1e 100644
--- a/docs/website/assets/css/extended/cn1-port-status.css
+++ b/docs/website/assets/css/extended/cn1-port-status.css
@@ -490,6 +490,21 @@
min-width: 1320px;
}
+.cn1-port-status__matrix--jdk25 table {
+ min-width: 900px;
+}
+
+.cn1-port-status__ratio {
+ display: block;
+ white-space: nowrap;
+}
+
+.cn1-port-status__ratio small {
+ color: var(--secondary);
+ display: block;
+ font-size: .68rem;
+}
+
.cn1-port-status__matrix--performance tbody th small {
color: var(--secondary);
display: block;
diff --git a/docs/website/layouts/_default/port-status.html b/docs/website/layouts/_default/port-status.html
index e98ae000d65..132241a93ab 100644
--- a/docs/website/layouts/_default/port-status.html
+++ b/docs/website/layouts/_default/port-status.html
@@ -319,6 +319,62 @@
+ {{- /* Rendered by scripts/website/build.sh from vm/selfhost/perf-baseline: the ratios
+ every platform build's performance gate holds ParparVM to, so this table and the
+ gate cannot disagree. Absent from a bare `hugo` run, hence the guard. */ -}}
+ {{- with site.Data.port_status_jdk25 }}
+ {{- $jdk := . -}}
+
How to read this page
diff --git a/scripts/website/build.sh b/scripts/website/build.sh
index 9309d857e91..10fcf5700cd 100755
--- a/scripts/website/build.sh
+++ b/scripts/website/build.sh
@@ -814,6 +814,13 @@ if [ "${WEBSITE_REFRESH_PORT_STATUS}" = "true" ]; then
"${REPO_ROOT}/scripts/website/sync_port_status_reports.sh"
fi
"${PYTHON_BIN}" "${REPO_ROOT}/scripts/hellocodenameone/conformance/port_status.py" validate
+# The ParparVM vs JDK 25 table is rendered from the performance gate's own baselines, the
+# ratios every platform build is held to. Generated here rather than committed: a copy in
+# the tree would go stale whenever a pull request rebaselines, and requiring each such
+# pull request to refresh it would bring back the merge conflicts the baseline layout
+# exists to avoid (vm/selfhost/perf_baseline.py).
+"${PYTHON_BIN}" "${REPO_ROOT}/vm/selfhost/perf_baseline.py" summary \
+ --out "${WEBSITE_DIR}/data/port_status_jdk25.json"
build_javadocs_for_site
build_developer_guide_for_site
diff --git a/scripts/website/preview.sh b/scripts/website/preview.sh
index ea569c43fee..5a7e288ca62 100755
--- a/scripts/website/preview.sh
+++ b/scripts/website/preview.sh
@@ -6,6 +6,9 @@ REPO_ROOT="$(cd "${SCRIPT_DIR}/../.." && pwd)"
WEBSITE_DIR="${REPO_ROOT}/docs/website"
python3 "${REPO_ROOT}/scripts/hellocodenameone/conformance/port_status.py" validate
+# Generated, never committed; see scripts/website/build.sh.
+python3 "${REPO_ROOT}/vm/selfhost/perf_baseline.py" summary \
+ --out "${WEBSITE_DIR}/data/port_status_jdk25.json"
if [ ! -d "${WEBSITE_DIR}" ]; then
echo "Website directory not found: ${WEBSITE_DIR}" >&2
diff --git a/scripts/website/validate_port_status.mjs b/scripts/website/validate_port_status.mjs
index 265bd648930..3def688f4e9 100644
--- a/scripts/website/validate_port_status.mjs
+++ b/scripts/website/validate_port_status.mjs
@@ -159,6 +159,26 @@ function validate() {
fail(`performance cells for ${name} must identify the platform in their tooltip`);
}
}
+ // The ParparVM / JDK 25 table, rendered from the performance gate's baselines by
+ // build.sh. Its size comes from the data it was rendered from, not a number spelled
+ // here, so adding a gated platform or benchmark cannot fail this one CI round at a time.
+ const jdkData = JSON.parse(fs.readFileSync(
+ path.resolve(publicDirectory, "..", "data", "port_status_jdk25.json"), "utf8"));
+ const jdkRows = countMatches(page, /\bdata-jdk25-row(?:=|\s|>)/g);
+ const jdkCellTags = page.match(/
]*data-jdk25-cell[^>]*>/g) || [];
+ if (jdkData.platforms.length < 5 || jdkRows !== jdkData.benchmarks.length ||
+ jdkCellTags.length !== jdkRows * jdkData.platforms.length) {
+ fail("the ParparVM against JDK 25 table is incomplete");
+ }
+ for (const platform of jdkData.platforms) {
+ if (!jdkCellTags.some((tag) => attribute(tag, "data-platform") === platform.id &&
+ attribute(tag, "title").startsWith(`${platform.name}:`))) {
+ fail(`JDK 25 cells for ${platform.name} must identify the platform in their tooltip`);
+ }
+ }
+ if (/Not measured/.test(page.match(/data-jdk25-evidence[\s\S]*?<\/section>/)?.[0] || "")) {
+ fail("a gated platform is missing a benchmark the gate measures everywhere");
+ }
const primaryCellTags = Array.from(page.matchAll(/ ]*\bdata-feature-cell\b)[^>]*>/gi), match => match[0]);
for (const cell of primaryCellTags) {
const port = attribute(cell, "data-port");
diff --git a/vm/selfhost/README.md b/vm/selfhost/README.md
index d01bc37818d..d71c0c4cba4 100644
--- a/vm/selfhost/README.md
+++ b/vm/selfhost/README.md
@@ -118,16 +118,33 @@ In a `--cores` sweep the count is pinned with CPU affinity on Linux and Windows.
has no affinity API, so there it is logical: both arms are told it (`CN1_GC_MARK_THREADS`,
`-XX:ActiveProcessorCount`) and neither is confined to it; the table marks those rows.
-**The gate compares against `perf-baseline.json`**: a ratio per platform and benchmark,
-and a tolerance per metric. A ratio more than the tolerance above its
-baseline fails the build, and the comment names the benchmark, the metric and the size
-of the change.
-
-**Calibrating.** Baselines must come from the runners that enforce them: a ratio depends
-on the hardware, so one measured on a developer machine is not a baseline for CI. A
-benchmark with no entry is reported as "not gated", and the comment carries the entry to
-add. When a change moves performance on purpose, update the entries in the same pull
-request, from that pull request's own run.
+**The gate compares against `perf-baseline/`**: a ratio per platform, runner CPU model
+and benchmark, and a tolerance per metric. A ratio more than the tolerance away from its
+baseline fails the build, in either direction -- above it is a regression, below it an
+improvement that has to be recorded, or a later change could give it back without
+failing anything. The comment names the benchmark, the metric and the size of the change.
+
+**Calibrating and rebaselining.** Baselines must come from the runners that enforce them:
+a ratio depends on the hardware, so one measured on a developer machine is not a baseline
+for CI. Both kinds of change go into the pull request's own
+`perf-baseline/pr/.json`, written from the failing job's `perf-results.json`:
+
+```bash
+# a runner CPU model with no rows yet
+python3 vm/selfhost/calibrate-perf-baseline.py --pr 5931 perf-results.json
+# a change that moves performance on purpose
+python3 vm/selfhost/calibrate-perf-baseline.py --pr 5931 --reason "why" perf-results.json
+```
+
+No pull request edits `perf-baseline/base/`. The nightly `fold` job in
+`.github/workflows/perf-baseline.yml` moves merged overlays there. The one-file layout
+this replaced made unrelated branches conflict on every merge; `perf_baseline.py`
+explains how overlays combine and when two of them are a real conflict. A branch still
+carrying edits to the old `perf-baseline.json` converts them with
+`perf_baseline.py import-legacy --pr N --reason "..." `.
+
+The Port Status page's ParparVM vs JDK 25 table is rendered from these same rows
+(`perf_baseline.py summary`, run by `scripts/website/build.sh`).
## Native collection and string implementation
diff --git a/vm/selfhost/calibrate-perf-baseline.py b/vm/selfhost/calibrate-perf-baseline.py
index c954a08706e..4c6fe0166e5 100644
--- a/vm/selfhost/calibrate-perf-baseline.py
+++ b/vm/selfhost/calibrate-perf-baseline.py
@@ -1,135 +1,193 @@
#!/usr/bin/env python3
-"""Writes perf-baseline.json from the perf-results.json of several CI runs of UNCHANGED code.
+"""Writes a pull request's baseline changes, pr/.json, from CI runs' perf-results.json.
- python3 vm/selfhost/calibrate-perf-baseline.py [--out perf-baseline.json] RESULTS.json ...
+ python3 vm/selfhost/calibrate-perf-baseline.py [--pr N] [--reason TEXT] [--all]
+ [--only BENCH,...] RESULTS.json ...
-Pass every perf-results.json you have -- several runs per platform, from the artifacts
+Pass the perf-results.json of the runs the gate failed on -- from the artifacts
ci-perf-gate.sh leaves (linux-screenshot-raw-*/perf, windows-port-screenshot-raw-*/perf,
-macos-ui-tests/perf). The runs must all measure the same code: the point is to learn how
-far each ratio moves when nothing changed.
-
-For every platform, benchmark and core setting:
+macos-ui-tests/perf). The rows land in vm/selfhost/perf-baseline/pr/.json, which only
+this pull request writes, so the change is in its diff and in nobody else's; the nightly
+fold moves it into base/ after the merge (see perf_baseline.py). --pr defaults to the pull
+request of the checked-out branch, asked of the GitHub CLI.
+
+What it writes, per platform, benchmark and core setting the runs measured:
+ calibrate for a row the baseline does not have (a runner CPU model no run had met).
+ rebaseline for a row the runs put outside its tolerance, either way: a change that
+ moved performance on purpose. Only the metric that moved is replaced, and
+ --reason is required -- the reason is what a reviewer reads beside the
+ number. --all rebaselines every measured row instead, for recalibrating a
+ row's noise from several runs of unchanged code.
+
+How a row is computed:
baseline the median of the runs' median ratios.
tolerance max(the global tolerance, SPREAD_MARGIN x the observed spread), where the spread
- is how far the runs' ratios reach above that median. A handful of runs
- under-samples the tail, hence the margin. Measured on this branch: across five
- runs most rows moved under 5%, while objectAllocation moved 20-37% on every
- platform and Windows x64's translation rows up to 48% -- one global tolerance
- either fails those on noise or waves real regressions through the rest.
-A row measured fewer than MIN_RUNS_FOR_OWN_SPREAD times cannot estimate its own spread;
-it takes the larger of what it measured and the widest tolerance any other row needed
-for the same benchmark and metric.
-
-Runs record the runner's CPU, and rows are written per CPU MODEL (`platform@model`, the
-model as perf-gate.cpu_class names it): hosted pools mix microarchitectures whose ratios
-differ by more than any tolerance absorbs. A runner on a model with no rows is reported
-and FAILS the gate until a run of it is calibrated in -- which is what this script is
-for: pass that run's perf-results.json and commit the result. A run whose CPU is unknown
-feeds the plain `platform` rows instead.
-
-Only the rows these results measure are replaced; every other row in the file is kept,
-so adding one new CPU model from one run leaves the rest alone. A single-run row borrows
-the widest tolerance seen for its benchmark across these runs AND the existing file.
---fresh starts from an empty set of rows instead: a full recalibration after the code
-changed, where a kept row would be a baseline for code that no longer exists.
-
-The global tolerance and the absolute floor (which keeps the tiny memory ratios, 0.03x of
-the JDK, from failing on a few kilobytes) are kept from the existing file.
+ is how far the runs' ratios reach from that median, in either direction --
+ an improvement fails the gate too, so a tolerance learned only from the high
+ side would fail unchanged code on its low runs. A handful of runs
+ under-samples the tail, hence the margin. Measured: across five runs most
+ rows moved under 5%, while objectAllocation moved 20-37% on every platform
+ and Windows x64's translation rows up to 48% -- one global tolerance either
+ fails those on noise or waves real regressions through the rest.
+A row measured fewer than MIN_RUNS_FOR_OWN_SPREAD times cannot estimate its own spread.
+A new row takes the larger of what it measured and the widest tolerance any row needed
+for the same benchmark and metric; a rebaselined row keeps at least its own.
+
+Rows are per CPU MODEL (`platform@model`, the model as perf-gate.cpu_class names it):
+hosted pools mix microarchitectures whose ratios differ by more than any tolerance
+absorbs. A run whose CPU is unknown feeds the plain `platform` rows instead.
"""
import argparse
+import copy
import importlib.util
import json
import math
import statistics
+import sys
from collections import defaultdict
from pathlib import Path
HERE = Path(__file__).resolve().parent
-_spec = importlib.util.spec_from_file_location('perf_gate', HERE / 'perf-gate.py')
-perf_gate = importlib.util.module_from_spec(_spec)
-_spec.loader.exec_module(perf_gate)
+
+
+def _load(name, file):
+ spec = importlib.util.spec_from_file_location(name, HERE / file)
+ module = importlib.util.module_from_spec(spec)
+ spec.loader.exec_module(module)
+ return module
+
+
+perf_gate = _load('perf_gate', 'perf-gate.py')
+perf_baseline = _load('perf_baseline', 'perf_baseline.py')
SPREAD_MARGIN = 1.5
# Fewer runs than this cannot estimate a row's spread: EPYC 7763's hello row, from two
# runs, failed unchanged code at +15.7% against a 15% tolerance. Such a row takes the
# widest tolerance its benchmark needed anywhere, as a single-run row always did.
MIN_RUNS_FOR_OWN_SPREAD = 5
-METRICS = ('time', 'memory')
+METRICS = perf_baseline.METRICS
def round_up(value, step=0.05):
return round(math.ceil(value / step - 1e-9) * step, 2)
-def main(argv=None):
- parser = argparse.ArgumentParser(description=__doc__.split('\n')[0])
- parser.add_argument('--out', default=str(HERE / 'perf-baseline.json'))
- parser.add_argument('--fresh', action='store_true',
- help='drop every row these results do not re-measure')
- parser.add_argument('results', nargs='+')
- args = parser.parse_args(argv)
+def spread_tolerance(values, base, floor):
+ """The tolerance a set of runs needs to pass themselves, or `floor` if smaller."""
+ if len(values) < 2:
+ return floor
+ spread = max(max(values) / base - 1, 1 - min(values) / base)
+ return max(floor, round_up(spread * SPREAD_MARGIN))
+
- existing = json.loads(Path(args.out).read_text())
- tolerance = existing['tolerance']
- # platform -> (benchmark, cores) -> metric -> [median ratio per run]
+def collect(paths, only=None):
+ """platform-key -> (benchmark, cores) -> metric -> [median ratio per run]."""
runs = defaultdict(lambda: defaultdict(lambda: defaultdict(list)))
- for path in args.results:
+ for path in paths:
report = json.loads(Path(path).read_text())
if report.get('error') or report.get('failures'):
raise SystemExit('%s did not complete; calibrate only from complete runs' % path)
cls = perf_gate.cpu_class(report.get('cpu'))
key = '%s@%s' % (report['platform'], cls) if cls else report['platform']
for bench, by_cores in report['results'].items():
+ if only and bench not in only:
+ continue
for cores, entry in by_cores.items():
for metric in METRICS:
runs[key][(bench, cores)][metric].append(entry[metric]['median'])
+ return runs
+
+
+def main(argv=None):
+ parser = argparse.ArgumentParser(description=__doc__.split('\n')[0])
+ parser.add_argument('--root', default=str(perf_baseline.ROOT))
+ parser.add_argument('--pr', type=int, help='the pull request number (default: ask gh)')
+ parser.add_argument('--reason', help='why the rebaselined rows moved')
+ parser.add_argument('--all', action='store_true',
+ help='rebaseline every measured row, not only those out of tolerance')
+ parser.add_argument('--only', help='comma-separated benchmark ids to take from the runs')
+ parser.add_argument('results', nargs='+')
+ args = parser.parse_args(argv)
- rows = {}
- widest = defaultdict(float) # (benchmark, metric) -> widest tolerance any platform needed
- kept = {} if args.fresh else {k: v for k, v in existing.get('platforms', {}).items()
- if k not in runs}
- for benches in kept.values():
- for bench, by_cores in benches.items():
- for row in by_cores.values():
- for metric, tol in row.get('tolerance', {}).items():
- widest[(bench, metric)] = max(widest[(bench, metric)], tol)
- for platform, benches in runs.items():
- for key, metrics in benches.items():
- row = {}
+ root = Path(args.root)
+ number = args.pr or perf_baseline.pr_number()
+ if number is None:
+ raise SystemExit('No pull request number: pass --pr N (the overlay is named after it)')
+ policy = perf_baseline.load_policy(root)
+ tolerance, floor = policy['tolerance'], policy['floor']
+ base = perf_baseline.load_base(root)
+ overlays = perf_baseline.load_overlays(root)
+ own = dict(overlays).get(number, {})
+ # What the gate judged against (this pull request's earlier rows included), and the
+ # baseline as it stands without them -- which is what a rebaseline's "from" names,
+ # since this run replaces this pull request's earlier rebaseline rather than stacking.
+ judged, _ = perf_baseline.resolve(base, overlays)
+ others, _ = perf_baseline.resolve(base, [o for o in overlays if o[0] != number])
+ runs = collect(args.results, set(args.only.split(',')) if args.only else None)
+
+ widest = defaultdict(float) # (benchmark, metric) -> widest tolerance any row has
+ for _, bench, _, row in perf_baseline._rows('baseline', judged):
+ for metric, tol in row.get('tolerance', {}).items():
+ widest[(bench, metric)] = max(widest[(bench, metric)], tol)
+ for benches in runs.values():
+ for (bench, _), metrics in benches.items():
for metric, values in metrics.items():
- base = statistics.median(values)
- row[metric] = round(base, 3)
+ own_tol = spread_tolerance(values, statistics.median(values), tolerance[metric])
if len(values) > 1:
- spread = max(values) / base - 1
- tol = max(tolerance[metric], round_up(spread * SPREAD_MARGIN))
- row.setdefault('tolerance', {})[metric] = tol
- widest[(key[0], metric)] = max(widest[(key[0], metric)], tol)
- rows[(platform, key)] = (row, min(len(v) for v in metrics.values()))
-
- platforms = {}
- for (platform, (bench, cores)), (row, count) in sorted(rows.items()):
- if count < MIN_RUNS_FOR_OWN_SPREAD:
- own = row.get('tolerance', {})
- row['tolerance'] = {m: max(tolerance[m], own.get(m, 0.0),
- widest.get((bench, m), tolerance[m]))
- for m in METRICS}
- # A row whose tolerance is just the global one does not need to repeat it.
- tol = {m: t for m, t in row.pop('tolerance', {}).items() if t > tolerance[m]}
- if tol:
- row['tolerance'] = tol
- row['runs'] = count
- platforms.setdefault(platform, {}).setdefault(bench, {})[cores] = row
-
- # Rows these results did not measure are kept (unless --fresh): a calibration that got
- # no macOS runner must not leave macOS without a baseline, and adding one CPU model
- # must not drop the platform's others.
- platforms.update(kept)
- existing['platforms'] = platforms
- Path(args.out).write_text(json.dumps(existing, indent=1, sort_keys=True) + '\n')
- for platform in sorted(platforms):
- print('%-14s %d benchmarks from %d run(s)' % (
- platform, len(platforms[platform]),
- max(r['runs'] for b in platforms[platform].values() for r in b.values())))
+ widest[(bench, metric)] = max(widest[(bench, metric)], own_tol)
+
+ calibrate, rebaseline = {}, {}
+ for key, benches in sorted(runs.items()):
+ for (bench, cores), metrics in sorted(benches.items()):
+ current = judged.get(key, {}).get(bench, {}).get(cores)
+ before = others.get(key, {}).get(bench, {}).get(cores)
+ new_row = current is None or cores in own.get('calibrate', {}).get(key, {}).get(bench, {})
+ row = {} if current is None else copy.deepcopy(current)
+ moved = []
+ for metric in METRICS:
+ values = metrics[metric]
+ if not (args.all or new_row or any(
+ perf_gate.verdict(v, current[metric],
+ current.get('tolerance', {}).get(metric, tolerance[metric]),
+ floor[metric]) != 'ok' for v in values)):
+ continue
+ moved.append(metric)
+ median = statistics.median(values)
+ row[metric] = round(median, 3)
+ tol = spread_tolerance(values, median, tolerance[metric])
+ if len(values) < MIN_RUNS_FOR_OWN_SPREAD:
+ kept = (current or {}).get('tolerance', {}).get(metric, 0.0)
+ tol = max(tol, kept, widest.get((bench, metric), 0.0) if new_row else 0.0)
+ row.setdefault('tolerance', {})[metric] = tol
+ if not moved:
+ continue
+ # A row whose tolerance is just the global one does not need to repeat it.
+ tol = {m: t for m, t in row.pop('tolerance', {}).items() if t > tolerance[m]}
+ if tol:
+ row['tolerance'] = tol
+ row['runs'] = min(len(metrics[m]) for m in moved)
+ if new_row or before is None:
+ calibrate.setdefault(key, {}).setdefault(bench, {})[cores] = row
+ else:
+ row['from'] = {m: before[m] for m in METRICS}
+ rebaseline.setdefault(key, {}).setdefault(bench, {})[cores] = row
+
+ if not calibrate and not rebaseline:
+ print('Every measured row is inside its tolerance; nothing to write.')
+ return 0
+ if rebaseline and not (args.reason or own.get('reason')):
+ raise SystemExit('These rows moved past their tolerance and would be rebaselined; say '
+ 'why with --reason:\n' + json.dumps(rebaseline, indent=1))
+ try:
+ path = perf_baseline.write_overlay(root, number, calibrate, rebaseline, args.reason)
+ perf_baseline.load(root) # the overlay must resolve against everything else
+ except perf_baseline.BaselineError as error:
+ raise SystemExit(str(error))
+ for kind, tree in (('calibrated', calibrate), ('rebaselined', rebaseline)):
+ for key, benches in sorted(tree.items()):
+ print('%s %s: %s' % (kind, key, ', '.join(sorted(benches))))
+ print('Wrote %s -- commit it with this pull request.' % path)
+ return 0
if __name__ == '__main__':
- main()
+ sys.exit(main())
diff --git a/vm/selfhost/ci-perf-gate.sh b/vm/selfhost/ci-perf-gate.sh
index ad99bab2d1d..fd61cace401 100755
--- a/vm/selfhost/ci-perf-gate.sh
+++ b/vm/selfhost/ci-perf-gate.sh
@@ -40,19 +40,29 @@ if report.get('failures'):
for f in report['failures']:
print(' %s at %s cores: %s' % (f['benchmark'], f['cores'], f['reason']))
sys.exit(1)
-if report.get('regression'):
+def moved(verdict):
lines = []
for bench, per_cores in report['results'].items():
for cores, entry in per_cores.items():
+ if 'failed' in entry:
+ continue
for metric in ('time', 'memory'):
e = entry[metric]
- if e['verdict'] == 'regression':
+ if e['verdict'] == verdict:
lines.append(' %s at %s cores: %s %.2fx against %.2fx (%+.1f%%)' % (
report['labels'].get(bench, bench), cores,
'time' if metric == 'time' else 'RAM', e['median'], e['baseline'],
(e['median'] / e['baseline'] - 1) * 100))
+ return lines
+number = report.get('pr') or ''
+overlay = 'vm/selfhost/perf-baseline/pr/%s.json' % number
+rebaseline = (' python3 vm/selfhost/calibrate-perf-baseline.py --pr %s --reason "" perf-results.json'
+ % number)
+if report.get('regression'):
print('ParparVM performance REGRESSION on %s:' % report['platform'])
- print('\n'.join(lines))
+ print('\n'.join(moved('regression')))
+ print('If this change costs performance on purpose, rebaseline it in %s:' % overlay)
+ print(rebaseline)
sys.exit(1)
missing = [(b, c) for b, per in report['results'].items() for c, e in per.items()
if 'failed' not in e and 'uncalibrated' in (e['time']['verdict'], e['memory']['verdict'])]
@@ -62,10 +72,18 @@ if missing or report.get('calibration'):
key = report.get('calibration_key') or report['platform']
print('ParparVM performance gate: NO BASELINE on %s for %s (runner CPU: %s).'
% (report['platform'], key, report.get('cpu', 'unknown')))
- print('Add it from this job\'s perf-results.json and commit perf-baseline.json:')
- print(' python3 vm/selfhost/calibrate-perf-baseline.py perf-results.json')
+ print('Add it to %s from this job\'s perf-results.json:' % overlay)
+ print(' python3 vm/selfhost/calibrate-perf-baseline.py --pr %s perf-results.json' % number)
print(json.dumps({key: report.get('calibration') or sorted(missing)}, indent=1))
sys.exit(1)
+improved = moved('improved')
+if improved:
+ # An improvement that is not written down lets a later change give it back unnoticed.
+ print('ParparVM performance IMPROVED past the baseline on %s:' % report['platform'])
+ print('\n'.join(improved))
+ print('Record the new baseline in %s:' % overlay)
+ print(rebaseline)
+ sys.exit(1)
print('ParparVM performance gate: no regression on %s' % report['platform'])
PY
fi
diff --git a/vm/selfhost/perf-baseline.json b/vm/selfhost/perf-baseline.json
deleted file mode 100644
index f10fced325b..00000000000
--- a/vm/selfhost/perf-baseline.json
+++ /dev/null
@@ -1,1930 +0,0 @@
-{
- "floor": {
- "memory": 0.05,
- "time": 0.0
- },
- "platforms": {
- "linux-arm64@neoverse-n2": {
- "arrayRandom": {
- "all": {
- "memory": 0.201,
- "runs": 7,
- "time": 0.937
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.367,
- "runs": 7,
- "time": 0.362,
- "tolerance": {
- "time": 0.25
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.121,
- "runs": 7,
- "time": 0.842
- }
- },
- "hello": {
- "all": {
- "memory": 0.868,
- "runs": 7,
- "time": 0.931
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.033,
- "runs": 7,
- "time": 1.045
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.033,
- "runs": 7,
- "time": 0.792
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.034,
- "runs": 7,
- "time": 1.098
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.239,
- "runs": 7,
- "time": 3.043,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.093,
- "runs": 7,
- "time": 0.993
- }
- },
- "recursion": {
- "all": {
- "memory": 0.035,
- "runs": 7,
- "time": 1.434
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.262,
- "runs": 7,
- "time": 1.36,
- "tolerance": {
- "memory": 0.3
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.542,
- "runs": 7,
- "time": 0.625
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.029,
- "runs": 7,
- "time": 0.508
- }
- }
- },
- "linux-x64@amd-epyc-7763-64-core-processor": {
- "arrayRandom": {
- "all": {
- "memory": 0.207,
- "runs": 3,
- "time": 0.91,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.36,
- "runs": 3,
- "time": 2.005,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.094,
- "runs": 3,
- "time": 1.483,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.836,
- "runs": 3,
- "time": 0.847,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.056,
- "runs": 3,
- "time": 1.116,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.054,
- "runs": 3,
- "time": 1.083,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.074,
- "runs": 3,
- "time": 1.084,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.37,
- "runs": 3,
- "time": 5.011,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.113,
- "runs": 3,
- "time": 1.073,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.076,
- "runs": 3,
- "time": 1.255,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.259,
- "runs": 3,
- "time": 1.39,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.519,
- "runs": 3,
- "time": 0.484,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.044,
- "runs": 3,
- "time": 0.102,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "linux-x64@amd-epyc-9v45-96-core-processor": {
- "arrayRandom": {
- "all": {
- "memory": 0.207,
- "runs": 1,
- "time": 1.023,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.356,
- "runs": 1,
- "time": 3.38,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.091,
- "runs": 1,
- "time": 1.206,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.86,
- "runs": 1,
- "time": 0.952,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.074,
- "runs": 1,
- "time": 0.999,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.074,
- "runs": 1,
- "time": 0.994,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.076,
- "runs": 1,
- "time": 1.192,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.363,
- "runs": 1,
- "time": 5.571,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.111,
- "runs": 1,
- "time": 1.056,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.075,
- "runs": 1,
- "time": 1.35,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.259,
- "runs": 1,
- "time": 1.557,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.539,
- "runs": 1,
- "time": 0.495,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.042,
- "runs": 1,
- "time": 0.101,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "linux-x64@amd-epyc-9v74-80-core-processor": {
- "arrayRandom": {
- "all": {
- "memory": 0.21,
- "runs": 2,
- "time": 1.024,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.355,
- "runs": 2,
- "time": 2.808,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.101,
- "runs": 2,
- "time": 1.096,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.786,
- "runs": 2,
- "time": 0.897,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.055,
- "runs": 2,
- "time": 1.103,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.055,
- "runs": 2,
- "time": 1.083,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.067,
- "runs": 2,
- "time": 1.093,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.388,
- "runs": 2,
- "time": 4.756,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.111,
- "runs": 2,
- "time": 1.069,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.056,
- "runs": 2,
- "time": 1.244,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.259,
- "runs": 2,
- "time": 1.489,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.522,
- "runs": 2,
- "time": 0.516,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.058,
- "runs": 2,
- "time": 0.101,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "linux-x64@intel-xeon-6973p-c": {
- "arrayRandom": {
- "all": {
- "memory": 0.207,
- "runs": 1,
- "time": 1.013,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.356,
- "runs": 1,
- "time": 2.128,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.098,
- "runs": 1,
- "time": 1.021,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.769,
- "runs": 1,
- "time": 0.982,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.054,
- "runs": 1,
- "time": 1.011,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.055,
- "runs": 1,
- "time": 1.066,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.057,
- "runs": 1,
- "time": 1.029,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.383,
- "runs": 1,
- "time": 5.399,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.128,
- "runs": 1,
- "time": 1.098,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.075,
- "runs": 1,
- "time": 1.301,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.221,
- "runs": 1,
- "time": 1.659,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.52,
- "runs": 1,
- "time": 0.537,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.048,
- "runs": 1,
- "time": 0.129,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "linux-x64@intel-xeon-platinum-8370c": {
- "arrayRandom": {
- "all": {
- "memory": 0.211,
- "runs": 3,
- "time": 1.011,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.355,
- "runs": 3,
- "time": 2.519,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.104,
- "runs": 3,
- "time": 1.094,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.782,
- "runs": 3,
- "time": 0.898,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.056,
- "runs": 3,
- "time": 1.031,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.054,
- "runs": 3,
- "time": 1.009,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.057,
- "runs": 3,
- "time": 1.029,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.504,
- "runs": 3,
- "time": 3.142,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.126,
- "runs": 3,
- "time": 1.057,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.075,
- "runs": 3,
- "time": 1.072,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.259,
- "runs": 3,
- "time": 1.559,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.509,
- "runs": 3,
- "time": 0.434,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.045,
- "runs": 3,
- "time": 0.133,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "linux-x64@intel-xeon-platinum-8573c": {
- "arrayRandom": {
- "all": {
- "memory": 0.206,
- "runs": 1,
- "time": 1.052,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.355,
- "runs": 1,
- "time": 2.693,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.096,
- "runs": 1,
- "time": 1.003,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.732,
- "runs": 1,
- "time": 0.942,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.074,
- "runs": 1,
- "time": 1.027,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.054,
- "runs": 1,
- "time": 0.962,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.057,
- "runs": 1,
- "time": 1.035,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.511,
- "runs": 1,
- "time": 2.631,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.11,
- "runs": 1,
- "time": 1.08,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.057,
- "runs": 1,
- "time": 1.055,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.221,
- "runs": 1,
- "time": 1.534,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.517,
- "runs": 1,
- "time": 0.505,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.044,
- "runs": 1,
- "time": 0.13,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "macos-arm64": {
- "arrayRandom": {
- "all": {
- "memory": 0.324,
- "runs": 1,
- "time": 0.989,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.527,
- "runs": 1,
- "time": 0.42,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.067,
- "runs": 1,
- "time": 1.176,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 1.34,
- "runs": 1,
- "time": 0.957,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.116,
- "runs": 1,
- "time": 1.033
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.112,
- "runs": 1,
- "time": 1.026,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.118,
- "runs": 1,
- "time": 1.007,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.481,
- "runs": 1,
- "time": 3.696,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.11,
- "runs": 1,
- "time": 1.012
- }
- },
- "recursion": {
- "all": {
- "memory": 0.118,
- "runs": 1,
- "time": 1.235,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.54,
- "runs": 1,
- "time": 0.842,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.739,
- "runs": 1,
- "time": 0.539
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.088,
- "runs": 1,
- "time": 0.507
- }
- }
- },
- "windows-arm64@armv8-64-bit-family-8-model-d49-microsoft-corporation": {
- "arrayRandom": {
- "all": {
- "memory": 0.228,
- "runs": 7,
- "time": 0.937
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.387,
- "runs": 7,
- "time": 0.394,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.119,
- "runs": 7,
- "time": 0.985
- }
- },
- "hello": {
- "all": {
- "memory": 0.9,
- "runs": 7,
- "time": 1.107
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.059,
- "runs": 7,
- "time": 1.044
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.059,
- "runs": 7,
- "time": 0.795
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.064,
- "runs": 7,
- "time": 0.72
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.463,
- "runs": 7,
- "time": 2.732
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.122,
- "runs": 7,
- "time": 1.001
- }
- },
- "recursion": {
- "all": {
- "memory": 0.065,
- "runs": 7,
- "time": 1.464
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.268,
- "runs": 7,
- "time": 1.427
- }
- },
- "translator": {
- "all": {
- "memory": 0.539,
- "runs": 7,
- "time": 0.751
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.05,
- "runs": 7,
- "time": 0.761
- }
- }
- },
- "windows-arm64@armv8-64-bit-family-8-model-d84-microsoft-corporation": {
- "arrayRandom": {
- "all": {
- "memory": 0.228,
- "runs": 1,
- "time": 0.941,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.386,
- "runs": 1,
- "time": 0.362,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.117,
- "runs": 1,
- "time": 0.814,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.696,
- "runs": 1,
- "time": 1.217,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.059,
- "runs": 1,
- "time": 0.997,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.059,
- "runs": 1,
- "time": 1.043,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.064,
- "runs": 1,
- "time": 0.627,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.284,
- "runs": 1,
- "time": 4.472,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.122,
- "runs": 1,
- "time": 0.935,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.064,
- "runs": 1,
- "time": 1.665,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.269,
- "runs": 1,
- "time": 1.527,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.57,
- "runs": 1,
- "time": 0.775,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.052,
- "runs": 1,
- "time": 0.762,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "windows-x64@amd64-family-25-model-1-authenticamd": {
- "arrayRandom": {
- "all": {
- "memory": 0.227,
- "runs": 7,
- "time": 1.061,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.385,
- "runs": 7,
- "time": 1.364
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.093,
- "runs": 7,
- "time": 1.798
- }
- },
- "hello": {
- "all": {
- "memory": 0.803,
- "runs": 7,
- "time": 1.241
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.056,
- "runs": 7,
- "time": 1.1
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.054,
- "runs": 7,
- "time": 1.081
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.059,
- "runs": 7,
- "time": 0.79
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.4,
- "runs": 7,
- "time": 4.932,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.12,
- "runs": 7,
- "time": 1.133
- }
- },
- "recursion": {
- "all": {
- "memory": 0.06,
- "runs": 7,
- "time": 1.494
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.322,
- "runs": 7,
- "time": 1.216
- }
- },
- "translator": {
- "all": {
- "memory": 0.542,
- "runs": 7,
- "time": 0.606
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.044,
- "runs": 7,
- "time": 0.1
- }
- }
- },
- "windows-x64@amd64-family-25-model-17-authenticamd": {
- "arrayRandom": {
- "all": {
- "memory": 0.226,
- "runs": 2,
- "time": 1.015,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.383,
- "runs": 2,
- "time": 1.718,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.097,
- "runs": 2,
- "time": 1.391,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.766,
- "runs": 2,
- "time": 1.375,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.056,
- "runs": 2,
- "time": 1.096,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.055,
- "runs": 2,
- "time": 1.081,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.058,
- "runs": 2,
- "time": 0.85,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.422,
- "runs": 2,
- "time": 6.089,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.12,
- "runs": 2,
- "time": 1.112,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.059,
- "runs": 2,
- "time": 1.428,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.278,
- "runs": 2,
- "time": 1.357,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.551,
- "runs": 2,
- "time": 0.636,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.045,
- "runs": 2,
- "time": 0.102,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "windows-x64@amd64-family-26-model-2-authenticamd": {
- "arrayRandom": {
- "all": {
- "memory": 0.226,
- "runs": 1,
- "time": 1.003,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.383,
- "runs": 1,
- "time": 1.747,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.089,
- "runs": 1,
- "time": 1.592,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.69,
- "runs": 1,
- "time": 1.41,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.054,
- "runs": 1,
- "time": 0.992,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.054,
- "runs": 1,
- "time": 0.989,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.059,
- "runs": 1,
- "time": 0.935,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.26,
- "runs": 1,
- "time": 5.86,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.119,
- "runs": 1,
- "time": 1.056,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.059,
- "runs": 1,
- "time": 1.441,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.291,
- "runs": 1,
- "time": 1.386,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.546,
- "runs": 1,
- "time": 0.611,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.045,
- "runs": 1,
- "time": 0.101,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "windows-x64@intel64-family-6-model-106-genuineintel": {
- "arrayRandom": {
- "all": {
- "memory": 0.226,
- "runs": 1,
- "time": 0.999,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.383,
- "runs": 1,
- "time": 1.425,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.103,
- "runs": 1,
- "time": 1.214,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.761,
- "runs": 1,
- "time": 1.212,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.055,
- "runs": 1,
- "time": 1.03,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.054,
- "runs": 1,
- "time": 1.002,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.059,
- "runs": 1,
- "time": 0.809,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.533,
- "runs": 1,
- "time": 3.284,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.119,
- "runs": 1,
- "time": 1.108,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.059,
- "runs": 1,
- "time": 1.262,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.27,
- "runs": 1,
- "time": 1.435,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.536,
- "runs": 1,
- "time": 0.627,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.044,
- "runs": 1,
- "time": 0.132,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- },
- "windows-x64@intel64-family-6-model-207-genuineintel": {
- "arrayRandom": {
- "all": {
- "memory": 0.226,
- "runs": 1,
- "time": 1.008,
- "tolerance": {
- "time": 0.45
- }
- }
- },
- "arraySequential": {
- "all": {
- "memory": 0.383,
- "runs": 1,
- "time": 1.982,
- "tolerance": {
- "time": 0.3
- }
- }
- },
- "hashMapChurn": {
- "all": {
- "memory": 0.098,
- "runs": 1,
- "time": 1.142,
- "tolerance": {
- "memory": 0.25,
- "time": 0.55
- }
- }
- },
- "hello": {
- "all": {
- "memory": 0.82,
- "runs": 1,
- "time": 1.069,
- "tolerance": {
- "memory": 0.2,
- "time": 0.5
- }
- }
- },
- "intArithmetic": {
- "all": {
- "memory": 0.059,
- "runs": 1,
- "time": 1.025,
- "tolerance": {
- "memory": 0.55
- }
- }
- },
- "longArithmetic": {
- "all": {
- "memory": 0.054,
- "runs": 1,
- "time": 0.971,
- "tolerance": {
- "memory": 0.5
- }
- }
- },
- "mathTranscendental": {
- "all": {
- "memory": 0.059,
- "runs": 1,
- "time": 0.886,
- "tolerance": {
- "memory": 0.4,
- "time": 0.3
- }
- }
- },
- "objectAllocation": {
- "all": {
- "memory": 0.562,
- "runs": 1,
- "time": 2.613,
- "tolerance": {
- "memory": 0.3,
- "time": 0.5
- }
- }
- },
- "quicksort": {
- "all": {
- "memory": 0.119,
- "runs": 1,
- "time": 1.066,
- "tolerance": {
- "memory": 0.25
- }
- }
- },
- "recursion": {
- "all": {
- "memory": 0.059,
- "runs": 1,
- "time": 1.204,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "stringBuilding": {
- "all": {
- "memory": 0.224,
- "runs": 1,
- "time": 1.415,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "translator": {
- "all": {
- "memory": 0.526,
- "runs": 1,
- "time": 0.602,
- "tolerance": {
- "time": 0.2
- }
- }
- },
- "valueEscape": {
- "all": {
- "memory": 0.045,
- "runs": 1,
- "time": 0.126,
- "tolerance": {
- "memory": 0.55
- }
- }
- }
- }
- },
- "tolerance": {
- "memory": 0.15,
- "time": 0.15
- }
-}
diff --git a/vm/selfhost/perf-baseline/README.md b/vm/selfhost/perf-baseline/README.md
new file mode 100644
index 00000000000..bf6994935fa
--- /dev/null
+++ b/vm/selfhost/perf-baseline/README.md
@@ -0,0 +1,11 @@
+# Performance gate baselines
+
+ParparVM / JDK 25 ratios that `vm/selfhost/perf-gate.py` holds every platform build to.
+
+- `policy.json`: the global tolerance per metric and the RAM floor.
+- `base/@.json`: the consolidated rows. Written only by the nightly
+ fold (`.github/workflows/perf-baseline.yml`); a pull request that edits it fails.
+- `pr/.json`: one pull request's calibrations and rebaselines, written by
+ `calibrate-perf-baseline.py --pr `. It is folded into `base/` after the merge.
+
+The format and the rules for combining overlays are in `../perf_baseline.py`.
diff --git a/vm/selfhost/perf-baseline/base/linux-arm64@neoverse-n2.json b/vm/selfhost/perf-baseline/base/linux-arm64@neoverse-n2.json
new file mode 100644
index 00000000000..0c2679912a5
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/linux-arm64@neoverse-n2.json
@@ -0,0 +1,102 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.201,
+ "runs": 7,
+ "time": 0.937
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.367,
+ "runs": 7,
+ "time": 0.362,
+ "tolerance": {
+ "time": 0.25
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.121,
+ "runs": 7,
+ "time": 0.842
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.868,
+ "runs": 7,
+ "time": 0.931
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.033,
+ "runs": 7,
+ "time": 1.045
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.033,
+ "runs": 7,
+ "time": 0.792
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.034,
+ "runs": 7,
+ "time": 1.098
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.239,
+ "runs": 7,
+ "time": 3.043,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.093,
+ "runs": 7,
+ "time": 0.993
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.035,
+ "runs": 7,
+ "time": 1.434
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.262,
+ "runs": 7,
+ "time": 1.36,
+ "tolerance": {
+ "memory": 0.3
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.542,
+ "runs": 7,
+ "time": 0.625
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.029,
+ "runs": 7,
+ "time": 0.508
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-7763-64-core-processor.json b/vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-7763-64-core-processor.json
new file mode 100644
index 00000000000..e17a1ad3957
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-7763-64-core-processor.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.207,
+ "runs": 3,
+ "time": 0.91,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.36,
+ "runs": 3,
+ "time": 2.005,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.094,
+ "runs": 3,
+ "time": 1.483,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.836,
+ "runs": 3,
+ "time": 0.847,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.056,
+ "runs": 3,
+ "time": 1.116,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.054,
+ "runs": 3,
+ "time": 1.083,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.074,
+ "runs": 3,
+ "time": 1.084,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.37,
+ "runs": 3,
+ "time": 5.011,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.113,
+ "runs": 3,
+ "time": 1.073,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.076,
+ "runs": 3,
+ "time": 1.255,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.259,
+ "runs": 3,
+ "time": 1.39,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.519,
+ "runs": 3,
+ "time": 0.484,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.044,
+ "runs": 3,
+ "time": 0.102,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v45-96-core-processor.json b/vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v45-96-core-processor.json
new file mode 100644
index 00000000000..4f55c2541fc
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v45-96-core-processor.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.207,
+ "runs": 1,
+ "time": 1.023,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.356,
+ "runs": 1,
+ "time": 3.38,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.091,
+ "runs": 1,
+ "time": 1.206,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.86,
+ "runs": 1,
+ "time": 0.952,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.074,
+ "runs": 1,
+ "time": 0.999,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.074,
+ "runs": 1,
+ "time": 0.994,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.076,
+ "runs": 1,
+ "time": 1.192,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.363,
+ "runs": 1,
+ "time": 5.571,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.111,
+ "runs": 1,
+ "time": 1.056,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.075,
+ "runs": 1,
+ "time": 1.35,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.259,
+ "runs": 1,
+ "time": 1.557,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.539,
+ "runs": 1,
+ "time": 0.495,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.042,
+ "runs": 1,
+ "time": 0.101,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v74-80-core-processor.json b/vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v74-80-core-processor.json
new file mode 100644
index 00000000000..56a4d1dfce8
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v74-80-core-processor.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.21,
+ "runs": 2,
+ "time": 1.024,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.355,
+ "runs": 2,
+ "time": 2.808,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.101,
+ "runs": 2,
+ "time": 1.096,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.786,
+ "runs": 2,
+ "time": 0.897,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.055,
+ "runs": 2,
+ "time": 1.103,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.055,
+ "runs": 2,
+ "time": 1.083,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.067,
+ "runs": 2,
+ "time": 1.093,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.388,
+ "runs": 2,
+ "time": 4.756,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.111,
+ "runs": 2,
+ "time": 1.069,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.056,
+ "runs": 2,
+ "time": 1.244,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.259,
+ "runs": 2,
+ "time": 1.489,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.522,
+ "runs": 2,
+ "time": 0.516,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.058,
+ "runs": 2,
+ "time": 0.101,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-6973p-c.json b/vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-6973p-c.json
new file mode 100644
index 00000000000..2071352b2f8
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-6973p-c.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.207,
+ "runs": 1,
+ "time": 1.013,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.356,
+ "runs": 1,
+ "time": 2.128,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.098,
+ "runs": 1,
+ "time": 1.021,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.769,
+ "runs": 1,
+ "time": 0.982,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.054,
+ "runs": 1,
+ "time": 1.011,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.055,
+ "runs": 1,
+ "time": 1.066,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.057,
+ "runs": 1,
+ "time": 1.029,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.383,
+ "runs": 1,
+ "time": 5.399,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.128,
+ "runs": 1,
+ "time": 1.098,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.075,
+ "runs": 1,
+ "time": 1.301,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.221,
+ "runs": 1,
+ "time": 1.659,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.52,
+ "runs": 1,
+ "time": 0.537,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.048,
+ "runs": 1,
+ "time": 0.129,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8370c.json b/vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8370c.json
new file mode 100644
index 00000000000..0b333f008d9
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8370c.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.211,
+ "runs": 3,
+ "time": 1.011,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.355,
+ "runs": 3,
+ "time": 2.519,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.104,
+ "runs": 3,
+ "time": 1.094,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.782,
+ "runs": 3,
+ "time": 0.898,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.056,
+ "runs": 3,
+ "time": 1.031,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.054,
+ "runs": 3,
+ "time": 1.009,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.057,
+ "runs": 3,
+ "time": 1.029,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.504,
+ "runs": 3,
+ "time": 3.142,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.126,
+ "runs": 3,
+ "time": 1.057,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.075,
+ "runs": 3,
+ "time": 1.072,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.259,
+ "runs": 3,
+ "time": 1.559,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.509,
+ "runs": 3,
+ "time": 0.434,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.045,
+ "runs": 3,
+ "time": 0.133,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8573c.json b/vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8573c.json
new file mode 100644
index 00000000000..a579b4a3149
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8573c.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.206,
+ "runs": 1,
+ "time": 1.052,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.355,
+ "runs": 1,
+ "time": 2.693,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.096,
+ "runs": 1,
+ "time": 1.003,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.732,
+ "runs": 1,
+ "time": 0.942,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.074,
+ "runs": 1,
+ "time": 1.027,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.054,
+ "runs": 1,
+ "time": 0.962,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.057,
+ "runs": 1,
+ "time": 1.035,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.511,
+ "runs": 1,
+ "time": 2.631,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.11,
+ "runs": 1,
+ "time": 1.08,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.057,
+ "runs": 1,
+ "time": 1.055,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.221,
+ "runs": 1,
+ "time": 1.534,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.517,
+ "runs": 1,
+ "time": 0.505,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.044,
+ "runs": 1,
+ "time": 0.13,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/macos-arm64.json b/vm/selfhost/perf-baseline/base/macos-arm64.json
new file mode 100644
index 00000000000..10b590d72d1
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/macos-arm64.json
@@ -0,0 +1,124 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.324,
+ "runs": 1,
+ "time": 0.989,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.527,
+ "runs": 1,
+ "time": 0.42,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.067,
+ "runs": 1,
+ "time": 1.176,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 1.34,
+ "runs": 1,
+ "time": 0.957,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.116,
+ "runs": 1,
+ "time": 1.033
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.112,
+ "runs": 1,
+ "time": 1.026,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.118,
+ "runs": 1,
+ "time": 1.007,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.481,
+ "runs": 1,
+ "time": 3.696,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.11,
+ "runs": 1,
+ "time": 1.012
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.118,
+ "runs": 1,
+ "time": 1.235,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.54,
+ "runs": 1,
+ "time": 0.842,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.739,
+ "runs": 1,
+ "time": 0.539
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.088,
+ "runs": 1,
+ "time": 0.507
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d49-microsoft-corporation.json b/vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d49-microsoft-corporation.json
new file mode 100644
index 00000000000..c23ed07f9e1
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d49-microsoft-corporation.json
@@ -0,0 +1,96 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.228,
+ "runs": 7,
+ "time": 0.937
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.387,
+ "runs": 7,
+ "time": 0.394,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.119,
+ "runs": 7,
+ "time": 0.985
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.9,
+ "runs": 7,
+ "time": 1.107
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.059,
+ "runs": 7,
+ "time": 1.044
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.059,
+ "runs": 7,
+ "time": 0.795
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.064,
+ "runs": 7,
+ "time": 0.72
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.463,
+ "runs": 7,
+ "time": 2.732
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.122,
+ "runs": 7,
+ "time": 1.001
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.065,
+ "runs": 7,
+ "time": 1.464
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.268,
+ "runs": 7,
+ "time": 1.427
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.539,
+ "runs": 7,
+ "time": 0.751
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.05,
+ "runs": 7,
+ "time": 0.761
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d84-microsoft-corporation.json b/vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d84-microsoft-corporation.json
new file mode 100644
index 00000000000..70b60742245
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d84-microsoft-corporation.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.228,
+ "runs": 1,
+ "time": 0.941,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.386,
+ "runs": 1,
+ "time": 0.362,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.117,
+ "runs": 1,
+ "time": 0.814,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.696,
+ "runs": 1,
+ "time": 1.217,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.059,
+ "runs": 1,
+ "time": 0.997,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.059,
+ "runs": 1,
+ "time": 1.043,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.064,
+ "runs": 1,
+ "time": 0.627,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.284,
+ "runs": 1,
+ "time": 4.472,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.122,
+ "runs": 1,
+ "time": 0.935,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.064,
+ "runs": 1,
+ "time": 1.665,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.269,
+ "runs": 1,
+ "time": 1.527,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.57,
+ "runs": 1,
+ "time": 0.775,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.052,
+ "runs": 1,
+ "time": 0.762,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-1-authenticamd.json b/vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-1-authenticamd.json
new file mode 100644
index 00000000000..72007384e2d
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-1-authenticamd.json
@@ -0,0 +1,100 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.227,
+ "runs": 7,
+ "time": 1.061,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.385,
+ "runs": 7,
+ "time": 1.364
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.093,
+ "runs": 7,
+ "time": 1.798
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.803,
+ "runs": 7,
+ "time": 1.241
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.056,
+ "runs": 7,
+ "time": 1.1
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.054,
+ "runs": 7,
+ "time": 1.081
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.059,
+ "runs": 7,
+ "time": 0.79
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.4,
+ "runs": 7,
+ "time": 4.932,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.12,
+ "runs": 7,
+ "time": 1.133
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.06,
+ "runs": 7,
+ "time": 1.494
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.322,
+ "runs": 7,
+ "time": 1.216
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.542,
+ "runs": 7,
+ "time": 0.606
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.044,
+ "runs": 7,
+ "time": 0.1
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-17-authenticamd.json b/vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-17-authenticamd.json
new file mode 100644
index 00000000000..cac3ebc5dc4
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-17-authenticamd.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.226,
+ "runs": 2,
+ "time": 1.015,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.383,
+ "runs": 2,
+ "time": 1.718,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.097,
+ "runs": 2,
+ "time": 1.391,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.766,
+ "runs": 2,
+ "time": 1.375,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.056,
+ "runs": 2,
+ "time": 1.096,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.055,
+ "runs": 2,
+ "time": 1.081,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.058,
+ "runs": 2,
+ "time": 0.85,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.422,
+ "runs": 2,
+ "time": 6.089,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.12,
+ "runs": 2,
+ "time": 1.112,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.059,
+ "runs": 2,
+ "time": 1.428,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.278,
+ "runs": 2,
+ "time": 1.357,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.551,
+ "runs": 2,
+ "time": 0.636,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.045,
+ "runs": 2,
+ "time": 0.102,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/windows-x64@amd64-family-26-model-2-authenticamd.json b/vm/selfhost/perf-baseline/base/windows-x64@amd64-family-26-model-2-authenticamd.json
new file mode 100644
index 00000000000..a1a2a1e87ac
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/windows-x64@amd64-family-26-model-2-authenticamd.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.226,
+ "runs": 1,
+ "time": 1.003,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.383,
+ "runs": 1,
+ "time": 1.747,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.089,
+ "runs": 1,
+ "time": 1.592,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.69,
+ "runs": 1,
+ "time": 1.41,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.054,
+ "runs": 1,
+ "time": 0.992,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.054,
+ "runs": 1,
+ "time": 0.989,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.059,
+ "runs": 1,
+ "time": 0.935,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.26,
+ "runs": 1,
+ "time": 5.86,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.119,
+ "runs": 1,
+ "time": 1.056,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.059,
+ "runs": 1,
+ "time": 1.441,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.291,
+ "runs": 1,
+ "time": 1.386,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.546,
+ "runs": 1,
+ "time": 0.611,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.045,
+ "runs": 1,
+ "time": 0.101,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-106-genuineintel.json b/vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-106-genuineintel.json
new file mode 100644
index 00000000000..5cd9deab15d
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-106-genuineintel.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.226,
+ "runs": 1,
+ "time": 0.999,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.383,
+ "runs": 1,
+ "time": 1.425,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.103,
+ "runs": 1,
+ "time": 1.214,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.761,
+ "runs": 1,
+ "time": 1.212,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.055,
+ "runs": 1,
+ "time": 1.03,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.054,
+ "runs": 1,
+ "time": 1.002,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.059,
+ "runs": 1,
+ "time": 0.809,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.533,
+ "runs": 1,
+ "time": 3.284,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.119,
+ "runs": 1,
+ "time": 1.108,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.059,
+ "runs": 1,
+ "time": 1.262,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.27,
+ "runs": 1,
+ "time": 1.435,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.536,
+ "runs": 1,
+ "time": 0.627,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.044,
+ "runs": 1,
+ "time": 0.132,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-207-genuineintel.json b/vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-207-genuineintel.json
new file mode 100644
index 00000000000..627300040c7
--- /dev/null
+++ b/vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-207-genuineintel.json
@@ -0,0 +1,136 @@
+{
+ "arrayRandom": {
+ "all": {
+ "memory": 0.226,
+ "runs": 1,
+ "time": 1.008,
+ "tolerance": {
+ "time": 0.45
+ }
+ }
+ },
+ "arraySequential": {
+ "all": {
+ "memory": 0.383,
+ "runs": 1,
+ "time": 1.982,
+ "tolerance": {
+ "time": 0.3
+ }
+ }
+ },
+ "hashMapChurn": {
+ "all": {
+ "memory": 0.098,
+ "runs": 1,
+ "time": 1.142,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.55
+ }
+ }
+ },
+ "hello": {
+ "all": {
+ "memory": 0.82,
+ "runs": 1,
+ "time": 1.069,
+ "tolerance": {
+ "memory": 0.2,
+ "time": 0.5
+ }
+ }
+ },
+ "intArithmetic": {
+ "all": {
+ "memory": 0.059,
+ "runs": 1,
+ "time": 1.025,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ },
+ "longArithmetic": {
+ "all": {
+ "memory": 0.054,
+ "runs": 1,
+ "time": 0.971,
+ "tolerance": {
+ "memory": 0.5
+ }
+ }
+ },
+ "mathTranscendental": {
+ "all": {
+ "memory": 0.059,
+ "runs": 1,
+ "time": 0.886,
+ "tolerance": {
+ "memory": 0.4,
+ "time": 0.3
+ }
+ }
+ },
+ "objectAllocation": {
+ "all": {
+ "memory": 0.562,
+ "runs": 1,
+ "time": 2.613,
+ "tolerance": {
+ "memory": 0.3,
+ "time": 0.5
+ }
+ }
+ },
+ "quicksort": {
+ "all": {
+ "memory": 0.119,
+ "runs": 1,
+ "time": 1.066,
+ "tolerance": {
+ "memory": 0.25
+ }
+ }
+ },
+ "recursion": {
+ "all": {
+ "memory": 0.059,
+ "runs": 1,
+ "time": 1.204,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "stringBuilding": {
+ "all": {
+ "memory": 0.224,
+ "runs": 1,
+ "time": 1.415,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "translator": {
+ "all": {
+ "memory": 0.526,
+ "runs": 1,
+ "time": 0.602,
+ "tolerance": {
+ "time": 0.2
+ }
+ }
+ },
+ "valueEscape": {
+ "all": {
+ "memory": 0.045,
+ "runs": 1,
+ "time": 0.126,
+ "tolerance": {
+ "memory": 0.55
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf-baseline/policy.json b/vm/selfhost/perf-baseline/policy.json
new file mode 100644
index 00000000000..1689d849f31
--- /dev/null
+++ b/vm/selfhost/perf-baseline/policy.json
@@ -0,0 +1,10 @@
+{
+ "floor": {
+ "memory": 0.05,
+ "time": 0.0
+ },
+ "tolerance": {
+ "memory": 0.15,
+ "time": 0.15
+ }
+}
diff --git a/vm/selfhost/perf-gate.py b/vm/selfhost/perf-gate.py
index 1fa68fc17ec..fb95a19b3a0 100644
--- a/vm/selfhost/perf-gate.py
+++ b/vm/selfhost/perf-gate.py
@@ -38,11 +38,14 @@
Every run is verified: translation output byte for byte against the first run, workload
checksums across arms and rounds. A ratio can never come from doing less work.
-A ratio more than the tolerance above its baseline in perf-baseline.json is a regression
-and the exit status is 1. So is a MISSING baseline -- a benchmark with no row, or a runner
+A ratio more than the tolerance above its baseline in vm/selfhost/perf-baseline/ is a
+regression and the exit status is 1. So is a ratio more than the tolerance BELOW it: an
+improvement that is not written down as the new baseline lets a later change give it back
+without failing anything. So is a MISSING baseline -- a benchmark with no row, or a runner
whose CPU model has none -- because a gate that goes quiet whenever it cannot judge stops
-preventing regressions without anyone noticing. The report carries the rows to add, and
-calibrate-perf-baseline.py adds them from the run's perf-results.json.
+preventing regressions without anyone noticing. Each of these is fixed in the pull
+request's own pr/.json, which calibrate-perf-baseline.py writes from the run's
+perf-results.json; perf_baseline.py explains the layout.
Core counts are enforced with CPU affinity on Linux and Windows (inherited by the child)
and by CN1_GC_MARK_THREADS plus -XX:ActiveProcessorCount everywhere. macOS has no affinity
@@ -70,6 +73,9 @@
_spec = importlib.util.spec_from_file_location('bench_selfhost', HERE / 'bench-selfhost.py')
bench = importlib.util.module_from_spec(_spec)
_spec.loader.exec_module(bench)
+_spec = importlib.util.spec_from_file_location('perf_baseline', HERE / 'perf_baseline.py')
+perf_baseline = importlib.util.module_from_spec(_spec)
+_spec.loader.exec_module(perf_baseline)
HELLO_APP = 'com_codenameone_examples_hellocodenameone_HelloCodenameOneStub'
HELLO_PKG = 'com.codenameone.examples.hellocodenameone'
@@ -147,7 +153,7 @@ def cpu_class(cpu):
def baseline_key(platforms, platform, cpu):
- """Which perf-baseline.json row set judges this runner, or None for none.
+ """Which perf-baseline/ row set judges this runner, or None for none.
platform@model when this CPU model was calibrated. When the platform HAS per-model
rows but none for this model, None: the runner is a microarchitecture no run has
@@ -372,12 +378,13 @@ def verdict(ratio, base, tolerance, floor=0.0):
"""A regression is a ratio more than `tolerance` above its baseline AND more than
`floor` above it in absolute terms. The floor is for RAM: a workload whose ParparVM
footprint is a few MB against the JVM's ~40MB sits at ratios near 0.05, where under a
- megabyte of jitter is a +30% change that means nothing."""
+ megabyte of jitter is a +30% change that means nothing. An improvement is the mirror
+ image, floor included, and FAILS the gate as well until it is rebaselined."""
if base is None:
return 'uncalibrated'
if ratio > base * (1 + tolerance) and ratio - base > floor:
return 'regression'
- if ratio < base * (1 - tolerance):
+ if ratio < base * (1 - tolerance) and base - ratio > floor:
return 'improved'
return 'ok'
@@ -425,11 +432,46 @@ def status_cell(entry):
verdicts = {entry['time']['verdict'], entry['memory']['verdict']}
if 'uncalibrated' in verdicts:
return '**NO BASELINE**'
- if 'improved' in verdicts:
- return 'better than baseline'
+ improved = ['%s %+.1f%%' % ('time' if m == 'time' else 'RAM',
+ (entry[m]['median'] / entry[m]['baseline'] - 1) * 100)
+ for m in ('time', 'memory') if entry[m]['verdict'] == 'improved']
+ if improved:
+ return '**IMPROVED: rebaseline** (%s)' % ', '.join(improved)
return 'ok'
+def _moved(report, verdict_):
+ """Prose lines for every row whose metric got `verdict_` ('regression'/'improved')."""
+ tol = report['tolerance']
+ lines = []
+ for bench_id, per_cores in report['results'].items():
+ for cores, entry in per_cores.items():
+ if 'failed' in entry:
+ continue
+ for metric in ('time', 'memory'):
+ e = entry[metric]
+ if e['verdict'] == verdict_:
+ lines.append('%s %s: %s %.2fx against a %.2fx baseline (%+.1f%%, '
+ 'tolerance %d%%)'
+ % (report['labels'][bench_id], _where(report, cores),
+ 'time' if metric == 'time' else 'RAM', e['median'],
+ e['baseline'], (e['median'] / e['baseline'] - 1) * 100,
+ round(e.get('tolerance', tol[metric]) * 100)))
+ return lines
+
+
+def overlay_name(report):
+ number = report.get('pr')
+ return 'vm/selfhost/perf-baseline/pr/%s.json' % (number if number else '')
+
+
+def fix_command(report, rebaseline):
+ number = report.get('pr')
+ return ('python3 vm/selfhost/calibrate-perf-baseline.py --pr %s%s perf-results.json'
+ % (number if number else '',
+ ' --reason ""' if rebaseline else ''))
+
+
def _where(report, key):
"""How a row's core count reads in prose."""
if key == 'all':
@@ -464,20 +506,8 @@ def render_markdown(report):
'see below'), '']
if report.get('error'):
lines += ['**The performance gate could not complete:** `%s`' % report['error'], '']
- regressions = []
- for bench_id, per_cores in report['results'].items():
- for cores, entry in per_cores.items():
- if 'failed' in entry:
- continue
- for metric in ('time', 'memory'):
- e = entry[metric]
- if e['verdict'] == 'regression':
- regressions.append('%s %s: %s %.2fx against a %.2fx baseline (%+.1f%%, '
- 'tolerance %d%%)'
- % (report['labels'][bench_id], _where(report, cores),
- 'time' if metric == 'time' else 'RAM', e['median'],
- e['baseline'], (e['median'] / e['baseline'] - 1) * 100,
- round(e.get('tolerance', tol[metric]) * 100)))
+ regressions = _moved(report, 'regression')
+ improvements = _moved(report, 'improved')
failures = report.get('failures') or []
if failures:
lines.append('**%d benchmark%s failed to run:**' % (len(failures),
@@ -490,12 +520,19 @@ def render_markdown(report):
'' if len(regressions) == 1 else 's'))
lines += ['- ' + r for r in regressions]
lines.append('')
+ if improvements:
+ lines.append('**%d improvement%s past the baseline, which fail%s until rebaselined:**'
+ % (len(improvements), '' if len(improvements) == 1 else 's',
+ 's' if len(improvements) == 1 else ''))
+ lines += ['- ' + r for r in improvements]
+ lines.append('')
lines += ['Ratios are **ParparVM / JDK 25**: below 1.00x ParparVM is faster (time) or '
'smaller (RAM). Median of %d interleaved, paired rounds; every run\'s output '
- 'was verified. A regression is a ratio more than %d%% (time) / %d%% (RAM) above '
- 'its baseline in `vm/selfhost/perf-baseline.json` (more, for a row whose '
- 'calibration runs were noisier; the file records it), and for RAM also more '
- 'than 0.05x above it in absolute terms.%s'
+ 'was verified. A ratio more than %d%% (time) / %d%% (RAM) away from its baseline in '
+ '`vm/selfhost/perf-baseline/` fails: above it is a regression, below it an '
+ 'improvement that has to be rebaselined (a row whose calibration runs were '
+ 'noisier carries a wider tolerance), and for RAM the change must also exceed '
+ '0.05x in absolute terms.%s'
% (report['rounds'], round(tol['time'] * 100), round(tol['memory'] * 100),
' Both run unpinned on all of the runner\'s CPUs, with their own default '
'thread counts.' if any('all' in per for per in report['results'].values())
@@ -522,21 +559,28 @@ def render_markdown(report):
+ '(`CN1_GC_MARK_THREADS`, `-XX:ActiveProcessorCount`) but neither is confined to it.']
if report.get('calibration'):
target = report.get('calibration_key') or report['platform']
- lines += ['', '**No baseline for %d row%s on `%s`, so this gate fails.** Add them from '
- 'this job\'s `perf-results.json` (in its uploaded artifact) and commit the '
- 'result:' % (sum(len(v) for v in report['calibration'].values()),
- '' if sum(len(v) for v in report['calibration'].values()) == 1
- else 's', target), '',
- '```', 'python3 vm/selfhost/calibrate-perf-baseline.py perf-results.json', '```',
+ count = sum(len(v) for v in report['calibration'].values())
+ lines += ['', '**No baseline for %d row%s on `%s`, so this gate fails.** Add %s to '
+ '`%s` from this job\'s `perf-results.json` (in its uploaded artifact) and '
+ 'commit it with this pull request:'
+ % (count, '' if count == 1 else 's', target, 'it' if count == 1 else 'them',
+ overlay_name(report)), '',
+ '```', fix_command(report, False), '```',
'', 'The rows it will add ', '',
'```json', json.dumps({target: report['calibration']}, indent=1),
'```', '', ' ']
+ if regressions or improvements:
+ lines += ['', 'A change that moves performance ON PURPOSE records the new baseline in '
+ '`%s`, with the reason, from this job\'s `perf-results.json`. The rebaseline '
+ 'is then part of this pull request\'s diff:' % overlay_name(report), '',
+ '```', fix_command(report, True), '```']
lines += ['', '**Result: %s**' % (
'performance regression' if report['regression'] else
('gate did not complete' if report.get('error') else
('benchmark failed' if report.get('failures') else
('no baseline: calibration required' if missing_baselines(report)
- else 'no regression'))))]
+ else ('improved past the baseline: rebaseline required' if improvements
+ else 'no regression')))))]
return '\n'.join(lines) + '\n'
@@ -556,7 +600,8 @@ def main(argv):
# allocator's steady state, where six processes agreed to 0.8%.
parser.add_argument('--reps', type=int, default=25,
help='measured repetitions per workload process (Bench argv[0])')
- parser.add_argument('--baseline', default=str(HERE / 'perf-baseline.json'))
+ parser.add_argument('--baseline', default=str(perf_baseline.ROOT),
+ help='the perf-baseline directory (base/, pr/, policy.json)')
parser.add_argument('--platform', default=platform_key())
parser.add_argument('--out', default=None)
parser.add_argument('--markdown', default=None)
@@ -573,7 +618,7 @@ def main(argv):
exe = '.exe' if platform.system() == 'Windows' else ''
report = {'platform': args.platform, 'rounds': args.rounds, 'results': {}, 'labels': {},
'available_cores': available_cores(), 'skipped_cores': [], 'regression': False,
- 'cpu': cpu_model()}
+ 'improved': False, 'cpu': cpu_model()}
out = Path(args.out) if args.out else None
markdown = Path(args.markdown) if args.markdown else None
@@ -585,8 +630,20 @@ def write():
markdown.parent.mkdir(parents=True, exist_ok=True)
markdown.write_text(render_markdown(report))
- baseline = json.loads(Path(args.baseline).read_text())
+ report['pr'] = perf_baseline.pr_number()
+ try:
+ baseline = perf_baseline.load(args.baseline)
+ except perf_baseline.BaselineError as error:
+ # Overlays that contradict each other leave nothing to judge against. Say which,
+ # in the comment, rather than measuring for ten minutes first.
+ report['tolerance'] = {'time': 0.0, 'memory': 0.0}
+ report['error'] = 'perf-baseline: %s' % error
+ write()
+ print('perf-gate: REFUSING: %s' % report['error'], flush=True)
+ return 2
report['tolerance'] = baseline['tolerance']
+ for note in baseline['notes']:
+ print('perf-gate: baseline note: %s' % note, flush=True)
try:
binary = Path(args.binary or os.environ.get('CN1_SELFHOST_BIN') or
TARGET / ('parpar-O3' + exe)).resolve()
@@ -676,6 +733,8 @@ def write():
floor))
if entry[metric]['verdict'] == 'regression':
report['regression'] = True
+ if entry[metric]['verdict'] == 'improved':
+ report['improved'] = True
report['results'].setdefault(spec['id'], {})[key] = entry
if base.get('time') is None or base.get('memory') is None:
calibration.setdefault(spec['id'], {})[key] = {
@@ -698,10 +757,12 @@ def write():
print(render_markdown(report))
failed = bool(report.get('failures'))
uncalibrated = bool(missing_baselines(report))
+ improved = bool(report.get('improved'))
print('perf-gate: %s' % ('REGRESSION' if report['regression'] else
('FAILED' if failed else
- ('NO BASELINE' if uncalibrated else 'OK'))))
- return 1 if report['regression'] or failed or uncalibrated else 0
+ ('NO BASELINE' if uncalibrated else
+ ('IMPROVED: REBASELINE' if improved else 'OK')))))
+ return 1 if report['regression'] or failed or uncalibrated or improved else 0
if __name__ == '__main__':
diff --git a/vm/selfhost/perf_baseline.py b/vm/selfhost/perf_baseline.py
new file mode 100644
index 00000000000..9a97e756480
--- /dev/null
+++ b/vm/selfhost/perf_baseline.py
@@ -0,0 +1,581 @@
+#!/usr/bin/env python3
+"""The performance gate's baselines: where they live, how a pull request changes them,
+and how those changes are folded in.
+
+ python3 vm/selfhost/perf_baseline.py check [--base REF] [--pr N]
+ python3 vm/selfhost/perf_baseline.py fold
+ python3 vm/selfhost/perf_baseline.py summary --out FILE
+ python3 vm/selfhost/perf_baseline.py import-legacy --pr N [--reason TEXT] --ref BRANCH
+
+THE LAYOUT (vm/selfhost/perf-baseline/)
+ policy.json the global tolerance per metric and the RAM floor
+ base/.json the consolidated rows of one runner, = platform@cpu-model
+ (or the bare platform where the CPU model is unknown):
+ {benchmark: {cores: {time, memory, runs, tolerance?}}}
+ pr/.json ONE pull request's changes to those rows
+
+Every row used to sit in a single perf-baseline.json, and every branch that met a new
+runner CPU or moved a benchmark edited it. Rows that have nothing to do with each other
+sat within git's three lines of context of each other, so independent branches
+conflicted, and two branches calibrating the same new CPU conflicted by construction.
+A pull request now writes only its own pr/.json -- a file no other branch can
+write -- and the change is still in that pull request's diff, which is the point of
+keeping baselines in the tree at all. base/ is written by `fold` alone, nightly on
+master, and a pull request that edits it fails `check`.
+
+AN OVERLAY (pr/.json)
+ {"pr": 5931,
+ "reason": "why any rebaselined row moved (required when there are any)",
+ "calibrate": {key: {benchmark: {cores: row}}},
+ "rebaseline": {key: {benchmark: {cores: row + "from": {"time": t, "memory": m}}}}}
+
+ calibrate a row that does not exist yet: a runner CPU model no run had measured.
+ Hardware, not code -- whichever branch met the runner first carries it.
+ Several overlays calibrating the same row are COMBINED (median of their
+ ratios, summed runs, the wider tolerance), so the result does not depend
+ on which merged first. A calibrate row whose row already exists is
+ superseded and ignored: once one branch's calibration is folded, a second
+ branch's adds nothing and must not fail for it.
+ rebaseline a row that exists and moved: a deliberate change in performance, better
+ or worse. "from" is the baseline the pull request measured against. If
+ the row is no longer at "from" -- another merged change moved it first --
+ the overlay is rejected, naming both: two changes moved the same
+ benchmark, and someone has to re-measure. That is the conflict git used
+ to report as a JSON hunk, now reported as what it is.
+
+Every overlay on master belongs to a merged pull request, so `fold` needs no knowledge
+of GitHub: it writes the resolved rows into base/, deletes every overlay, and refuses to
+commit unless the resolved baseline is identical before and after -- the fold changes no
+verdict, by construction and by check.
+"""
+import argparse
+import copy
+import json
+import math
+import os
+from pathlib import Path
+import re
+import statistics
+import subprocess
+import sys
+
+HERE = Path(__file__).resolve().parent
+ROOT = HERE / 'perf-baseline'
+REPO = HERE.parents[1]
+METRICS = ('time', 'memory')
+KEY_RE = re.compile(r'^[a-z0-9]+-[a-z0-9]+(@[a-z0-9-]+)?$')
+ROW_FIELDS = {'time', 'memory', 'runs', 'tolerance'}
+# As the Port Status page spells them (scripts/website/validate_port_status.mjs).
+PLATFORM_NAMES = {'linux-x64': 'Linux x64', 'linux-arm64': 'Linux ARM64',
+ 'macos-arm64': 'macOS ARM64', 'macos-x64': 'macOS x64',
+ 'windows-x64': 'Windows x64', 'windows-arm64': 'Windows ARM64'}
+PLATFORM_ORDER = ['linux-x64', 'linux-arm64', 'windows-x64', 'windows-arm64',
+ 'macos-arm64', 'macos-x64']
+# What the Port Status page says about each benchmark. The workloads are
+# vm/benchmarks/common's CommonWorkloads, the same code the page's absolute table runs
+# inside each port's application, so these descriptions match
+# docs/website/data/port_status_support.json's (test_perf_gate holds them in step).
+BENCHMARKS = [
+ ('hello', 'Translating an application',
+ 'The ParparVM translator translating the compliance test application, exactly as '
+ 'the platform build translates it.'),
+ ('translator', 'Translating the translator',
+ 'The ParparVM translator translating its own classes, the Java API and ASM.'),
+ ('intArithmetic', 'Integer arithmetic',
+ '40 million dependent 32-bit multiply, add, shift, and XOR operations.'),
+ ('longArithmetic', 'Long arithmetic',
+ '30 million dependent 64-bit arithmetic and bitwise operations.'),
+ ('mathTranscendental', 'Transcendental math',
+ 'Eight million data-dependent sqrt, sin, cos, and remainder operations.'),
+ ('arraySequential', 'Sequential arrays',
+ 'Fill eight million integers and perform four complete reduction passes.'),
+ ('arrayRandom', 'Random array access',
+ '20 million data-dependent reads from a four-million-element array.'),
+ ('objectAllocation', 'Object allocation',
+ 'Allocate eight million short-lived linked nodes with periodic traversal.'),
+ ('valueEscape', 'Non-escaping value objects',
+ 'Eight million two-field value objects that never escape the loop, the shape escape '
+ 'analysis turns into registers.'),
+ ('hashMapChurn', 'Hash map churn',
+ 'Three million boxed-key lookups and updates with repeated table clearing.'),
+ ('stringBuilding', 'String building',
+ 'Build, retain, and hash 400,000 data-dependent strings.'),
+ ('recursion', 'Recursion',
+ 'Three recursive Fibonacci calls using inputs 35 and 36.'),
+ ('quicksort', 'Quicksort',
+ 'Generate and sort 1.5 million integers, then verify ordering and checksum.'),
+]
+
+
+class BaselineError(Exception):
+ """A baseline the gate cannot judge against. Always fatal: a gate that skips what it
+ cannot read stops preventing regressions without anyone noticing."""
+
+
+def _read_json(path):
+ try:
+ return json.loads(path.read_text())
+ except ValueError as error:
+ raise BaselineError('%s is not valid JSON: %s' % (path, error))
+
+
+def dump(value):
+ """The one serialization every file here uses, so a rewrite of unchanged data is a
+ byte-for-byte no-op."""
+ return json.dumps(value, indent=1, sort_keys=True) + '\n'
+
+
+def _check_row(where, row, extra=()):
+ unknown = set(row) - ROW_FIELDS - set(extra)
+ if unknown:
+ raise BaselineError('%s: unknown field(s) %s' % (where, ', '.join(sorted(unknown))))
+ for metric in METRICS:
+ value = row.get(metric)
+ if isinstance(value, bool) or not isinstance(value, (int, float)) or not value > 0:
+ raise BaselineError('%s: %s must be a positive ratio, not %r' % (where, metric, value))
+ runs = row.get('runs')
+ if isinstance(runs, bool) or not isinstance(runs, int) or runs < 1:
+ raise BaselineError('%s: runs must be a positive integer, not %r' % (where, runs))
+ tolerance = row.get('tolerance', {})
+ if not isinstance(tolerance, dict) or set(tolerance) - set(METRICS):
+ raise BaselineError('%s: tolerance must map time/memory to a fraction' % where)
+ for metric, value in tolerance.items():
+ if isinstance(value, bool) or not isinstance(value, (int, float)) or not 0 < value < 5:
+ raise BaselineError('%s: %s tolerance %r is not a fraction' % (where, metric, value))
+
+
+def _rows(where, tree):
+ """(key, benchmark, cores, row) for every row of a {key: {bench: {cores: row}}} tree."""
+ if not isinstance(tree, dict):
+ raise BaselineError('%s must be an object' % where)
+ for key, benches in sorted(tree.items()):
+ if not KEY_RE.match(key):
+ raise BaselineError('%s: %r is not a platform or platform@cpu-model key' % (where, key))
+ if not isinstance(benches, dict):
+ raise BaselineError('%s: %s must be an object' % (where, key))
+ for bench, per_cores in sorted(benches.items()):
+ if not isinstance(per_cores, dict):
+ raise BaselineError('%s: %s/%s must be an object' % (where, key, bench))
+ for cores, row in sorted(per_cores.items()):
+ if not isinstance(row, dict):
+ raise BaselineError('%s: %s/%s/%s must be an object' % (where, key, bench, cores))
+ yield key, bench, cores, row
+
+
+def load_policy(root=ROOT):
+ policy = _read_json(Path(root) / 'policy.json')
+ for name in ('tolerance', 'floor'):
+ if set(policy.get(name, {})) != set(METRICS):
+ raise BaselineError('policy.json: %s must give time and memory' % name)
+ return policy
+
+
+def load_base(root=ROOT):
+ base = {}
+ directory = Path(root) / 'base'
+ for path in sorted(directory.glob('*')):
+ if path.suffix != '.json':
+ raise BaselineError('%s: only .json files belong in base/' % path)
+ key = path.stem
+ for _, bench, cores, row in _rows(str(path), {key: _read_json(path)}):
+ _check_row('%s %s/%s' % (path.name, bench, cores), row)
+ base.setdefault(key, {}).setdefault(bench, {})[cores] = row
+ return base
+
+
+def load_overlays(root=ROOT):
+ overlays = []
+ for path in sorted(Path(root).joinpath('pr').glob('*')):
+ if not re.match(r'^[1-9][0-9]*\.json$', path.name):
+ raise BaselineError('%s: an overlay is named .json' % path)
+ overlay = _read_json(path)
+ validate_overlay(int(path.stem), overlay, path.name)
+ overlays.append((int(path.stem), overlay))
+ return sorted(overlays)
+
+
+def validate_overlay(number, overlay, where):
+ if not isinstance(overlay, dict):
+ raise BaselineError('%s must be an object' % where)
+ unknown = set(overlay) - {'pr', 'reason', 'calibrate', 'rebaseline'}
+ if unknown:
+ raise BaselineError('%s: unknown field(s) %s' % (where, ', '.join(sorted(unknown))))
+ if overlay.get('pr') != number:
+ raise BaselineError('%s: "pr" must be %d, the number in its file name' % (where, number))
+ for _, bench, cores, row in _rows(where + ' calibrate', overlay.get('calibrate', {})):
+ _check_row('%s calibrate %s/%s' % (where, bench, cores), row)
+ rebaselines = list(_rows(where + ' rebaseline', overlay.get('rebaseline', {})))
+ for key, bench, cores, row in rebaselines:
+ here = '%s rebaseline %s %s/%s' % (where, key, bench, cores)
+ _check_row(here, row, extra=('from',))
+ old = row.get('from')
+ if not isinstance(old, dict) or set(old) != set(METRICS):
+ raise BaselineError('%s: "from" must give the time and memory baseline it '
+ 'replaces' % here)
+ if rebaselines:
+ reason = overlay.get('reason')
+ if not isinstance(reason, str) or not reason.strip() or reason.strip().upper().startswith('TODO'):
+ raise BaselineError('%s: a rebaseline needs a "reason" saying why the rows moved'
+ % where)
+ if not overlay.get('calibrate') and not rebaselines:
+ raise BaselineError('%s changes nothing; delete it' % where)
+
+
+def _same(a, b):
+ return all(math.isclose(a[m], b[m], rel_tol=0, abs_tol=5e-4) for m in METRICS)
+
+
+def _combine(rows):
+ """Several branches calibrating the same new row: one calibration from all of them."""
+ if len(rows) == 1:
+ return copy.deepcopy(rows[0])
+ combined = {m: round(statistics.median(r[m] for r in rows), 3) for m in METRICS}
+ combined['runs'] = sum(r['runs'] for r in rows)
+ tolerance = {}
+ for r in rows:
+ for metric, value in r.get('tolerance', {}).items():
+ tolerance[metric] = max(tolerance.get(metric, 0.0), value)
+ if tolerance:
+ combined['tolerance'] = tolerance
+ return combined
+
+
+def resolve(base, overlays):
+ """The rows the gate judges against: base with every overlay applied.
+
+ Returns (rows, notes). Raises BaselineError for overlays that contradict each other or
+ the base, naming the pull requests involved."""
+ rows = copy.deepcopy(base)
+ notes = []
+ calibrations = {}
+ for number, overlay in overlays:
+ for key, bench, cores, row in _rows('pr/%d.json' % number, overlay.get('calibrate', {})):
+ calibrations.setdefault((key, bench, cores), []).append((number, row))
+ for (key, bench, cores), entries in sorted(calibrations.items()):
+ if cores in rows.get(key, {}).get(bench, {}):
+ notes.append('%s %s/%s: already calibrated; the calibration in %s is superseded'
+ % (key, bench, cores, ', '.join('pr/%d.json' % n for n, _ in entries)))
+ continue
+ rows.setdefault(key, {}).setdefault(bench, {})[cores] = _combine([r for _, r in entries])
+ rebaselines = {}
+ for number, overlay in overlays:
+ for key, bench, cores, row in _rows('pr/%d.json' % number, overlay.get('rebaseline', {})):
+ rebaselines.setdefault((key, bench, cores), []).append((number, row))
+ for (key, bench, cores), entries in sorted(rebaselines.items()):
+ where = '%s %s/%s' % (key, bench, cores)
+ if len(entries) > 1:
+ raise BaselineError(
+ '%s is rebaselined by more than one pull request (%s): two changes moved the '
+ 'same benchmark. Re-measure on top of both and keep one rebaseline.'
+ % (where, ', '.join('pr/%d.json' % n for n, _ in entries)))
+ number, row = entries[0]
+ current = rows.get(key, {}).get(bench, {}).get(cores)
+ if current is None:
+ raise BaselineError('pr/%d.json rebaselines %s, which has no row; a new row is a '
+ '"calibrate" entry' % (number, where))
+ if not _same(row['from'], current):
+ raise BaselineError(
+ 'pr/%d.json rebaselines %s from time %.3fx / RAM %.3fx, but the row is now '
+ 'time %.3fx / RAM %.3fx: another merged change moved it first. Re-measure '
+ 'on top of it and update the rebaseline.'
+ % (number, where, row['from']['time'], row['from']['memory'],
+ current['time'], current['memory']))
+ rows[key][bench][cores] = {k: v for k, v in row.items() if k != 'from'}
+ return rows, notes
+
+
+def load(root=ROOT):
+ """Everything perf-gate.py needs: tolerance, floor and the resolved rows."""
+ policy = load_policy(root)
+ rows, notes = resolve(load_base(root), load_overlays(root))
+ return {'tolerance': policy['tolerance'], 'floor': policy['floor'], 'platforms': rows,
+ 'notes': notes}
+
+
+def write_base(root, rows):
+ directory = Path(root) / 'base'
+ directory.mkdir(parents=True, exist_ok=True)
+ wanted = {'%s.json' % key for key in rows}
+ for path in directory.glob('*.json'):
+ if path.name not in wanted:
+ path.unlink()
+ for key, benches in rows.items():
+ (directory / ('%s.json' % key)).write_text(dump(benches))
+
+
+def fold(root=ROOT):
+ """Write the resolved rows into base/ and delete the overlays. Returns the overlays
+ folded. Verifies the gate reads the same baseline afterwards."""
+ root = Path(root)
+ overlays = load_overlays(root)
+ before, _ = resolve(load_base(root), overlays)
+ if not overlays:
+ return []
+ write_base(root, before)
+ for number, _ in overlays:
+ (root / 'pr' / ('%d.json' % number)).unlink()
+ after, _ = resolve(load_base(root), load_overlays(root))
+ if after != before:
+ raise BaselineError('folding changed the resolved baseline; nothing may be committed')
+ return [number for number, _ in overlays]
+
+
+def write_overlay(root, number, calibrate=None, rebaseline=None, reason=None):
+ """Add rows to pr/.json, creating it if needed. A row written again replaces
+ the earlier one, so re-running the calibration after another CI round is safe."""
+ path = Path(root) / 'pr' / ('%d.json' % number)
+ overlay = _read_json(path) if path.exists() else {'pr': number}
+ for kind, tree in (('calibrate', calibrate), ('rebaseline', rebaseline)):
+ for key, bench, cores, row in _rows(kind, tree or {}):
+ overlay.setdefault(kind, {}).setdefault(key, {}).setdefault(bench, {})[cores] = row
+ # One row is one kind: a row calibrated earlier on this branch and now moved
+ # again is still a calibration, and a stale entry of the other kind would
+ # contradict it.
+ other = overlay.get('rebaseline' if kind == 'calibrate' else 'calibrate', {})
+ other.get(key, {}).get(bench, {}).pop(cores, None)
+ for kind in ('calibrate', 'rebaseline'):
+ tree = overlay.get(kind, {})
+ for key in list(tree):
+ for bench in list(tree[key]):
+ if not tree[key][bench]:
+ del tree[key][bench]
+ if not tree[key]:
+ del tree[key]
+ if kind in overlay and not tree:
+ del overlay[kind]
+ if reason:
+ overlay['reason'] = reason
+ validate_overlay(number, overlay, path.name)
+ path.parent.mkdir(parents=True, exist_ok=True)
+ path.write_text(dump(overlay))
+ return path
+
+
+def pr_number():
+ """This pull request's number, from the CI event or the GitHub CLI; None when unknown."""
+ explicit = os.environ.get('CN1_PR_NUMBER', '')
+ if explicit.isdigit():
+ return int(explicit)
+ event = os.environ.get('GITHUB_EVENT_PATH')
+ if event and Path(event).is_file():
+ try:
+ number = (json.loads(Path(event).read_text()).get('pull_request') or {}).get('number')
+ if number:
+ return int(number)
+ except (ValueError, OSError):
+ pass
+ match = re.match(r'^refs/pull/(\d+)/', os.environ.get('GITHUB_REF', ''))
+ if match:
+ return int(match.group(1))
+ # Most of these workflows are run by workflow_dispatch on a branch, which carries no
+ # pull request; ask GitHub which pull request the branch belongs to.
+ branch = os.environ.get('GITHUB_REF_NAME') if os.environ.get('GITHUB_ACTIONS') else None
+ try:
+ out = subprocess.run(['gh', 'pr', 'view'] + ([branch] if branch else []) +
+ ['--json', 'number', '-q', '.number'],
+ capture_output=True, text=True, timeout=30, cwd=str(REPO))
+ if out.returncode == 0 and out.stdout.strip().isdigit():
+ return int(out.stdout.strip())
+ except (OSError, subprocess.SubprocessError):
+ pass
+ return None
+
+
+def platform_of(key):
+ return key.split('@', 1)[0]
+
+
+def summary(rows):
+ """The Port Status page's ParparVM vs JDK 25 table: per platform and benchmark, the
+ median ratio across the platform's calibrated CPU models and the range they span.
+ Only the default all-cores rows are published."""
+ platforms = {}
+ for key, benches in rows.items():
+ platforms.setdefault(platform_of(key), {})[key] = benches
+ order = [p for p in PLATFORM_ORDER if p in platforms] + \
+ sorted(p for p in platforms if p not in PLATFORM_ORDER)
+ out = []
+ for platform in order:
+ keys = platforms[platform]
+ entry = {'id': platform, 'name': PLATFORM_NAMES.get(platform, platform),
+ 'cpus': sorted(k.split('@', 1)[1] if '@' in k else 'unidentified CPU'
+ for k in keys),
+ 'benchmarks': {}}
+ for bench, _, _ in BENCHMARKS:
+ values = [benches[bench]['all'] for benches in keys.values()
+ if 'all' in benches.get(bench, {})]
+ if not values:
+ continue
+ entry['benchmarks'][bench] = {
+ metric: {'median': round(statistics.median(v[metric] for v in values), 3),
+ 'min': min(v[metric] for v in values),
+ 'max': max(v[metric] for v in values)}
+ for metric in METRICS}
+ out.append(entry)
+ return {'schema_version': 1,
+ 'source': 'vm/selfhost/perf-baseline',
+ 'benchmarks': [{'id': b, 'name': n, 'description': d} for b, n, d in BENCHMARKS],
+ 'platforms': out}
+
+
+def changed_files(base_ref, root=ROOT):
+ rel = Path(root).resolve().relative_to(REPO).as_posix()
+ out = subprocess.run(['git', 'diff', '--name-only', '--no-renames', '%s...HEAD' % base_ref,
+ '--', rel], capture_output=True, text=True, cwd=str(REPO))
+ if out.returncode != 0:
+ raise BaselineError('git diff against %s failed: %s' % (base_ref, out.stderr.strip()))
+ return [line[len(rel) + 1:] for line in out.stdout.splitlines() if line.strip()]
+
+
+def base_exists_at(ref, root=ROOT):
+ rel = Path(root).resolve().relative_to(REPO).as_posix()
+ return subprocess.run(['git', 'cat-file', '-e', '%s:%s/base' % (ref, rel)],
+ capture_output=True, cwd=str(REPO)).returncode == 0
+
+
+def check(root=ROOT, base_ref=None, number=None):
+ """Problems with the baselines as a list of strings; empty when the gate can use them.
+ With base_ref, also what a pull request may change: its own overlay and the policy,
+ never base/ (fold's alone) and never another pull request's overlay."""
+ problems = []
+ try:
+ data = load(root)
+ for note in data['notes']:
+ print('note: ' + note)
+ except BaselineError as error:
+ problems.append(str(error))
+ if base_ref:
+ try:
+ changed = changed_files(base_ref, root)
+ except BaselineError as error:
+ return problems + [str(error)]
+ # The change that introduced this layout creates base/; that is the one pull
+ # request allowed to, and it is recognisable by base/ not existing before it.
+ migrating = not base_exists_at(base_ref, root)
+ for path in changed:
+ if path.startswith('base/') and not migrating:
+ problems.append('%s changed: base/ is written only by the nightly fold. Put the '
+ 'rows in pr/.json instead '
+ '(calibrate-perf-baseline.py --pr N writes it).' % path)
+ elif path.startswith('pr/') and number is not None and path != 'pr/%d.json' % number:
+ problems.append('%s changed: a pull request writes only its own overlay, '
+ 'pr/%d.json' % (path, number))
+ return problems
+
+
+LEGACY_FILE = 'vm/selfhost/perf-baseline.json'
+
+
+def _git_show(ref, path):
+ out = subprocess.run(['git', 'show', '%s:%s' % (ref, path)], capture_output=True,
+ text=True, cwd=str(REPO))
+ if out.returncode != 0:
+ raise BaselineError('%s has no %s' % (ref, path))
+ return json.loads(out.stdout)
+
+
+def legacy_from_ref(ref):
+ """(the branch's perf-baseline.json, the one it branched from). Only the difference
+ between the two is the branch's own: a branch that predates a master commit to the old
+ file carries master's PREVIOUS value for that row, which is not an edit of its own."""
+ out = subprocess.run(['git', 'merge-base', ref, 'HEAD'], capture_output=True, text=True,
+ cwd=str(REPO))
+ if out.returncode != 0:
+ raise BaselineError('no merge base between %s and HEAD' % ref)
+ try:
+ original = _git_show(out.stdout.strip(), LEGACY_FILE)
+ except BaselineError:
+ original = {'platforms': {}} # the branch created the file: every row is its own
+ return _git_show(ref, LEGACY_FILE), original
+
+
+def import_legacy(root, number, legacy, original=None, reason=None):
+ """Turn a branch's edits to the old single perf-baseline.json into its overlay. `legacy`
+ is the branch's copy, `original` the copy it branched from; only rows that differ
+ between the two are the branch's. Rows the resolved baseline lacks become calibrations,
+ rows it has become rebaselines from their current value. Returns (path or None, notes)."""
+ current = load(root)['platforms']
+ before = (original or {}).get('platforms')
+ calibrate, rebaseline, notes = {}, {}, []
+ for key, bench, cores, row in _rows('the branch perf-baseline.json', legacy.get('platforms', {})):
+ _check_row('%s %s/%s' % (key, bench, cores), row)
+ if before is not None and before.get(key, {}).get(bench, {}).get(cores) == row:
+ continue # not this branch's edit
+ existing = current.get(key, {}).get(bench, {}).get(cores)
+ old = (before or {}).get(key, {}).get(bench, {}).get(cores)
+ if existing is None:
+ calibrate.setdefault(key, {}).setdefault(bench, {})[cores] = row
+ elif before is not None and old is None:
+ # The branch calibrated a row master has calibrated since, from its own runs:
+ # two calibrations of one CPU, and the one already in the tree stands.
+ notes.append('%s %s/%s: calibrated on master too; master\'s row stands'
+ % (key, bench, cores))
+ elif existing != row:
+ if before is not None and old is not None and old != existing:
+ notes.append('%s %s/%s was also changed on master since the branch point; '
+ 'check the imported value' % (key, bench, cores))
+ moved = dict(row, **{'from': {m: existing[m] for m in METRICS}})
+ rebaseline.setdefault(key, {}).setdefault(bench, {})[cores] = moved
+ if not calibrate and not rebaseline:
+ return None, notes
+ return write_overlay(root, number, calibrate, rebaseline, reason), notes
+
+
+def main(argv=None):
+ parser = argparse.ArgumentParser(description=__doc__.split('\n')[0])
+ parser.add_argument('--root', default=str(ROOT))
+ sub = parser.add_subparsers(dest='command', required=True)
+ p = sub.add_parser('check', help='validate the baselines (and, with --base, a PR diff)')
+ p.add_argument('--base', help='the pull request base commit')
+ p.add_argument('--pr', type=int, help='the pull request number')
+ sub.add_parser('fold', help='fold every overlay into base/ (nightly, on master)')
+ p = sub.add_parser('summary', help='write the Port Status JDK 25 table data')
+ p.add_argument('--out', required=True)
+ p = sub.add_parser('import-legacy', help="convert a branch's perf-baseline.json edits")
+ p.add_argument('--pr', type=int)
+ p.add_argument('--reason')
+ p.add_argument('--ref', help='the branch, read from git with its merge base (preferred)')
+ p.add_argument('--legacy', help="the branch's perf-baseline.json, if not using --ref")
+ p.add_argument('--original', help='the perf-baseline.json the branch started from')
+ args = parser.parse_args(argv)
+ root = Path(args.root)
+ try:
+ if args.command == 'check':
+ problems = check(root, args.base, args.pr)
+ for problem in problems:
+ print('perf-baseline: ' + problem)
+ if problems:
+ return 1
+ print('perf-baseline: OK')
+ elif args.command == 'fold':
+ folded = fold(root)
+ print('perf-baseline: folded %s' % (', '.join('pr/%d.json' % n for n in folded)
+ if folded else 'nothing'))
+ elif args.command == 'summary':
+ out = Path(args.out)
+ out.parent.mkdir(parents=True, exist_ok=True)
+ out.write_text(dump(summary(load(root)['platforms'])))
+ print('perf-baseline: wrote %s' % out)
+ elif args.command == 'import-legacy':
+ number = args.pr or pr_number()
+ if number is None:
+ raise BaselineError('no pull request number: pass --pr')
+ if args.ref:
+ legacy, original = legacy_from_ref(args.ref)
+ elif args.legacy:
+ legacy = _read_json(Path(args.legacy))
+ original = _read_json(Path(args.original)) if args.original else None
+ else:
+ raise BaselineError('pass --ref BRANCH, or --legacy FILE [--original FILE]')
+ path, notes = import_legacy(root, number, legacy, original, args.reason)
+ for note in notes:
+ print('perf-baseline: note: ' + note)
+ print('perf-baseline: %s' % (path or "the branch changed no baseline row"))
+ except BaselineError as error:
+ print('perf-baseline: ' + str(error))
+ return 1
+ return 0
+
+
+if __name__ == '__main__':
+ sys.exit(main())
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index 03a67de7596..a2be5453e3c 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -38,6 +38,10 @@ def test_tolerance_is_relative_to_the_baseline(self):
self.assertEqual('improved', gate.verdict(0.89, 1.0, 0.1))
self.assertEqual('uncalibrated', gate.verdict(5.0, None, 0.1))
+ def test_an_improvement_inside_the_floor_is_not_one(self):
+ self.assertEqual('ok', gate.verdict(0.04, 0.06, 0.15, 0.05))
+ self.assertEqual('improved', gate.verdict(0.40, 0.60, 0.15, 0.05))
+
def test_a_ram_change_below_the_floor_is_not_a_regression(self):
# 0.06x -> 0.08x is +33%, and under a megabyte for a few-MB process.
self.assertEqual('ok', gate.verdict(0.08, 0.06, 0.15, 0.05))
@@ -87,6 +91,24 @@ def test_an_uncalibrated_cpu_fails_and_says_how_to_fix_it(self):
self.assertIn('**no baseline for this CPU model** -- the gate fails', text)
self.assertIn('**Result: no baseline: calibration required**', text)
+ def test_an_improvement_fails_and_says_how_to_rebaseline(self):
+ r = report({'quicksort': {'all': (metric(0.7, 1.0, 'improved'), metric(0.9, 0.9, 'ok'))}})
+ r['pr'] = 5931
+ text = gate.render_markdown(r)
+ self.assertIn('**IMPROVED: rebaseline** (time -30.0%)', text)
+ self.assertIn('vm/selfhost/perf-baseline/pr/5931.json', text)
+ self.assertIn('--pr 5931 --reason', text)
+ self.assertIn('**Result: improved past the baseline: rebaseline required**', text)
+
+ def test_a_calibration_names_this_pull_requests_overlay(self):
+ r = report({'quicksort': {'all': (metric(1.3, None, 'uncalibrated'),
+ metric(0.9, None, 'uncalibrated'))}})
+ r['calibration'] = {'quicksort': {'all': {'time': 1.3, 'memory': 0.9}}}
+ r['calibration_key'] = 'linux-x64@new'
+ text = gate.render_markdown(r)
+ self.assertIn('pr/.json', text)
+ self.assertIn('calibrate-perf-baseline.py --pr perf-results.json', text)
+
def test_an_incomplete_gate_says_so(self):
text = gate.render_markdown(report({}, error='stale native build'))
self.assertIn('could not complete', text)
@@ -118,19 +140,49 @@ def test_only_ratios_are_printed(self):
calibrate = load('calibrate_perf_baseline', 'calibrate-perf-baseline.py')
+baselines = load('perf_baseline', 'perf_baseline.py')
+POLICY = {'tolerance': {'time': 0.15, 'memory': 0.15}, 'floor': {'time': 0.0, 'memory': 0.05}}
-class CalibrationTest(unittest.TestCase):
- """calibrate-perf-baseline.py: medians as baselines, spread-driven tolerances."""
+class BaselineTree:
+ """A throwaway perf-baseline directory: policy.json, base/, pr/."""
- def run_calibration(self, runs, existing=None, fresh=False):
+ def __init__(self, base=None, overlays=None):
import json
import tempfile
- with tempfile.TemporaryDirectory() as tmp:
- out = Path(tmp) / 'baseline.json'
- out.write_text(json.dumps({'tolerance': {'time': 0.15, 'memory': 0.15},
- 'floor': {'time': 0.0, 'memory': 0.05},
- 'platforms': existing or {}}))
+ self._tmp = tempfile.TemporaryDirectory()
+ self.root = Path(self._tmp.name) / 'perf-baseline'
+ (self.root / 'pr').mkdir(parents=True)
+ (self.root / 'policy.json').write_text(json.dumps(POLICY))
+ baselines.write_base(self.root, base or {})
+ for number, overlay in (overlays or {}).items():
+ (self.root / 'pr' / ('%d.json' % number)).write_text(json.dumps(overlay))
+
+ def overlay(self, number):
+ import json
+ path = self.root / 'pr' / ('%d.json' % number)
+ return json.loads(path.read_text()) if path.exists() else None
+
+ def close(self):
+ self._tmp.cleanup()
+
+
+def row(t, m, runs=3, **tolerance):
+ r = {'time': t, 'memory': m, 'runs': runs}
+ if tolerance:
+ r['tolerance'] = tolerance
+ return r
+
+
+class CalibrationTest(unittest.TestCase):
+ """calibrate-perf-baseline.py: medians as baselines, spread-driven tolerances, written
+ into the pull request's own overlay."""
+
+ def run_calibration(self, runs, existing=None, overlays=None, pr=7, reason=None,
+ everything=False):
+ import json
+ tree = BaselineTree(existing, overlays)
+ try:
files = []
for i, run in enumerate(runs):
platform, per_bench = run[0], run[1]
@@ -139,49 +191,62 @@ def run_calibration(self, runs, existing=None, fresh=False):
report = {'platform': platform, 'results': results}
if len(run) > 2:
report['cpu'] = run[2]
- f = Path(tmp) / ('run%d.json' % i)
+ f = tree.root.parent / ('run%d.json' % i)
f.write_text(json.dumps(report))
files.append(str(f))
- calibrate.main(['--out', str(out)] + (['--fresh'] if fresh else []) + files)
- return json.loads(out.read_text())
+ calibrate.main(['--root', str(tree.root), '--pr', str(pr)] +
+ (['--reason', reason] if reason else []) +
+ (['--all'] if everything else []) + files)
+ return tree.overlay(pr), baselines.load(tree.root)['platforms']
+ finally:
+ tree.close()
def test_median_baseline_and_global_tolerance_for_a_steady_row(self):
- b = self.run_calibration([('linux-x64', {'quicksort': (1.00, 0.10)}),
- ('linux-x64', {'quicksort': (1.02, 0.10)}),
- ('linux-x64', {'quicksort': (1.01, 0.10)})])
- row = b['platforms']['linux-x64']['quicksort']['all']
- self.assertEqual(row['time'], 1.01)
- self.assertNotIn('tolerance', row) # 1% spread: the global 15% already covers it
- self.assertEqual(row['runs'], 3)
-
- def test_noisy_row_gets_a_wider_tolerance_that_covers_every_run(self):
- runs = [('linux-x64', {'objectAllocation': (v, 0.4)}) for v in (5.0, 5.5, 6.5)]
- b = self.run_calibration(runs)
- row = b['platforms']['linux-x64']['objectAllocation']['all']
- self.assertEqual(row['time'], 5.5)
- tol = row['tolerance']['time']
+ overlay, rows = self.run_calibration([('linux-x64', {'quicksort': (1.00, 0.10)}),
+ ('linux-x64', {'quicksort': (1.02, 0.10)}),
+ ('linux-x64', {'quicksort': (1.01, 0.10)})])
+ r = overlay['calibrate']['linux-x64']['quicksort']['all']
+ self.assertEqual(r['time'], 1.01)
+ self.assertNotIn('tolerance', r) # 1% spread: the global 15% already covers it
+ self.assertEqual(r['runs'], 3)
+ self.assertEqual(rows['linux-x64']['quicksort']['all'], r)
+
+ def test_noisy_row_gets_a_wider_tolerance_that_covers_every_run_both_ways(self):
+ runs = [('linux-x64', {'objectAllocation': (v, 0.4)}) for v in (3.5, 5.5, 6.5)]
+ overlay, _ = self.run_calibration(runs)
+ r = overlay['calibrate']['linux-x64']['objectAllocation']['all']
+ self.assertEqual(r['time'], 5.5)
+ tol = r['tolerance']['time']
self.assertGreater(tol, 0.15)
+ # An improvement fails too, so the LOW run (3.5, -36%) must pass as well.
for _, per in runs:
- self.assertNotEqual(gate.verdict(per['objectAllocation'][0], row['time'], tol), 'regression')
+ self.assertEqual(gate.verdict(per['objectAllocation'][0], r['time'], tol), 'ok')
# ...and it still bites: well past the observed spread is a regression.
- self.assertEqual(gate.verdict(row['time'] * (1 + tol) * 1.01, row['time'], tol), 'regression')
+ self.assertEqual(gate.verdict(r['time'] * (1 + tol) * 1.01, r['time'], tol), 'regression')
- def test_single_run_platform_borrows_the_widest_tolerance_seen_elsewhere(self):
+ def test_single_run_row_borrows_the_widest_tolerance_seen_elsewhere(self):
runs = [('linux-x64', {'objectAllocation': (v, 0.4)}) for v in (5.0, 5.5, 6.5)]
runs.append(('macos-arm64', {'objectAllocation': (3.0, 0.3)}))
- b = self.run_calibration(runs)
- linux = b['platforms']['linux-x64']['objectAllocation']['all']['tolerance']['time']
- mac = b['platforms']['macos-arm64']['objectAllocation']['all']
+ overlay, _ = self.run_calibration(runs)
+ linux = overlay['calibrate']['linux-x64']['objectAllocation']['all']['tolerance']['time']
+ mac = overlay['calibrate']['macos-arm64']['objectAllocation']['all']
self.assertEqual(mac['runs'], 1)
self.assertEqual(mac['tolerance']['time'], linux)
+ def test_a_single_run_row_borrows_tolerance_from_the_baseline(self):
+ old = {'windows-x64@a': {'arraySequential': {'all': row(1.35, 0.4, time=0.7)}}}
+ overlay, _ = self.run_calibration(
+ [('windows-x64', {'arraySequential': (1.67, 0.38)}, 'Model B')], existing=old)
+ self.assertEqual(overlay['calibrate']['windows-x64@model-b']['arraySequential']['all']
+ ['tolerance']['time'], 0.7)
+
def test_rows_are_per_cpu_model_and_an_unseen_model_is_not_gated(self):
zen3 = 'AMD64 Family 25 Model 1 Stepping 1, AuthenticAMD'
zen4 = 'AMD64 Family 25 Model 17 Stepping 1, AuthenticAMD'
intel = 'Intel64 Family 6 Model 207 Stepping 2, GenuineIntel'
runs = [('windows-x64', {'arraySequential': (v, 0.4)}, zen3) for v in (1.35, 1.36, 1.38)]
runs += [('windows-x64', {'arraySequential': (1.98, 0.38)}, intel)]
- b = self.run_calibration(runs)['platforms']
+ _, b = self.run_calibration(runs)
self.assertEqual(sorted(b), ['windows-x64@amd64-family-25-model-1-authenticamd',
'windows-x64@intel64-family-6-model-207-genuineintel'])
self.assertEqual(gate.baseline_key(b, 'windows-x64', zen3.replace('Stepping 1', 'Stepping 2')),
@@ -191,39 +256,246 @@ def test_rows_are_per_cpu_model_and_an_unseen_model_is_not_gated(self):
self.assertIsNone(gate.baseline_key(b, 'windows-x64', None))
def test_a_run_with_no_known_cpu_feeds_the_plain_row(self):
- b = self.run_calibration([('macos-arm64', {'quicksort': (1.0, 0.1)})])['platforms']
+ _, b = self.run_calibration([('macos-arm64', {'quicksort': (1.0, 0.1)})])
self.assertEqual(sorted(b), ['macos-arm64'])
self.assertEqual(gate.baseline_key(b, 'macos-arm64', 'Apple M1 (Virtual)'), 'macos-arm64')
- def test_only_measured_rows_are_replaced_unless_fresh(self):
- old = {'macos-arm64': {'quicksort': {'all': {'time': 0.9, 'memory': 0.1, 'runs': 1}}},
- 'linux-x64': {'quicksort': {'all': {'time': 9.9, 'memory': 0.9, 'runs': 1}}},
- 'linux-x64@intel': {'quicksort': {'all': {'time': 9.9, 'memory': 0.9, 'runs': 1}}}}
- b = self.run_calibration([('linux-x64', {'quicksort': (1.0, 0.1)})], existing=old)
- b = b['platforms']
- self.assertEqual(b['macos-arm64'], old['macos-arm64'])
- self.assertEqual(b['linux-x64']['quicksort']['all']['time'], 1.0)
- # Adding one CPU model's rows must not drop the platform's other models.
- self.assertEqual(b['linux-x64@intel'], old['linux-x64@intel'])
- fresh = self.run_calibration([('linux-x64', {'quicksort': (1.0, 0.1)})], existing=old,
- fresh=True)['platforms']
- self.assertEqual(sorted(fresh), ['linux-x64'])
-
- def test_a_thinly_sampled_row_borrows_the_widest_tolerance(self):
- runs = [('linux-x64', {'hello': (v, 0.8)}, 'CPU A') for v in (0.50, 0.70, 0.60, 0.55, 0.65)]
- runs += [('linux-x64', {'hello': (v, 0.8)}, 'CPU B') for v in (0.79, 0.80)]
- b = self.run_calibration(runs)['platforms']
- wide = b['linux-x64@cpu-a']['hello']['all']['tolerance']['time']
- self.assertEqual(b['linux-x64@cpu-a']['hello']['all']['runs'], 5)
- # Two runs 1% apart would give 15%; two runs cannot estimate a spread.
- self.assertEqual(b['linux-x64@cpu-b']['hello']['all']['tolerance']['time'], wide)
-
- def test_a_single_run_row_borrows_tolerance_from_the_existing_file(self):
- old = {'windows-x64@a': {'arraySequential': {'all': {
- 'time': 1.35, 'memory': 0.4, 'runs': 3, 'tolerance': {'time': 0.7}}}}}
- b = self.run_calibration([('windows-x64', {'arraySequential': (1.67, 0.38)}, 'Model B')],
- existing=old)['platforms']
- self.assertEqual(b['windows-x64@model-b']['arraySequential']['all']['tolerance']['time'], 0.7)
+ def test_a_row_inside_its_tolerance_is_left_alone(self):
+ old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1)},
+ 'recursion': {'all': row(1.0, 0.1)}}}
+ overlay, _ = self.run_calibration(
+ [('linux-x64', {'quicksort': (1.05, 0.1), 'recursion': (1.4, 0.1)})],
+ existing=old, reason='recursion got slower on purpose')
+ self.assertEqual(sorted(overlay['rebaseline']['linux-x64']), ['recursion'])
+ self.assertNotIn('calibrate', overlay)
+
+ def test_a_moved_row_is_rebaselined_from_its_old_value_and_needs_a_reason(self):
+ old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1, memory=0.3)}}}
+ with self.assertRaises(SystemExit):
+ self.run_calibration([('linux-x64', {'quicksort': (0.7, 0.1)})], existing=old)
+ overlay, rows = self.run_calibration([('linux-x64', {'quicksort': (0.7, 0.1)})],
+ existing=old, reason='faster partitioning')
+ r = overlay['rebaseline']['linux-x64']['quicksort']['all']
+ self.assertEqual(r['from'], {'time': 1.0, 'memory': 0.1})
+ self.assertEqual(r['time'], 0.7)
+ # Only the metric that moved is replaced; RAM keeps its baseline and tolerance.
+ self.assertEqual(r['memory'], 0.1)
+ self.assertEqual(r['tolerance'], {'memory': 0.3})
+ self.assertEqual(overlay['reason'], 'faster partitioning')
+ self.assertEqual(rows['linux-x64']['quicksort']['all']['time'], 0.7)
+
+ def test_all_recalibrates_every_measured_row(self):
+ old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1)}}}
+ overlay, _ = self.run_calibration(
+ [('linux-x64', {'quicksort': (v, 0.1)}) for v in (0.97, 1.0, 1.02, 1.01, 0.99)],
+ existing=old, reason='five runs of unchanged code', everything=True)
+ r = overlay['rebaseline']['linux-x64']['quicksort']['all']
+ self.assertEqual((r['time'], r['runs']), (1.0, 5))
+
+ def test_rerunning_replaces_this_pull_requests_own_rows(self):
+ old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1)}}}
+ mine = {'pr': 7, 'reason': 'first try', 'rebaseline': {'linux-x64': {'quicksort': {
+ 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}})}}}}
+ overlay, rows = self.run_calibration([('linux-x64', {'quicksort': (0.5, 0.1)})],
+ existing=old, overlays={7: mine})
+ r = overlay['rebaseline']['linux-x64']['quicksort']['all']
+ # Still FROM the base value, not from this pull request's own earlier 0.7.
+ self.assertEqual(r['from']['time'], 1.0)
+ self.assertEqual(rows['linux-x64']['quicksort']['all']['time'], 0.5)
+
+ def test_a_row_this_pull_request_calibrated_stays_a_calibration(self):
+ mine = {'pr': 7, 'calibrate': {'linux-x64@new': {'quicksort': {'all': row(1.0, 0.1, 1)}}}}
+ overlay, _ = self.run_calibration([('linux-x64', {'quicksort': (1.5, 0.1)}, 'New')],
+ overlays={7: mine})
+ self.assertNotIn('rebaseline', overlay)
+ self.assertEqual(overlay['calibrate']['linux-x64@new']['quicksort']['all']['time'], 1.5)
+
+
+class OverlayTests(unittest.TestCase):
+ """perf_baseline.py: how base/ and the pull requests' overlays become one baseline."""
+
+ BASE = {'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1)}}}
+
+ def resolve(self, overlays, base=None):
+ for number, overlay in overlays:
+ baselines.validate_overlay(number, overlay, 'pr/%d.json' % number)
+ return baselines.resolve(base if base is not None else self.BASE, overlays)
+
+ def rebase(self, number, t, frm=1.0, reason='moved'):
+ return (number, {'pr': number, 'reason': reason, 'rebaseline': {'linux-x64@a': {
+ 'quicksort': {'all': dict(row(t, 0.1), **{'from': {'time': frm, 'memory': 0.1}})}}}})
+
+ def calib(self, number, t, key='linux-x64@b', runs=1, **tol):
+ return (number, {'pr': number, 'calibrate': {key: {'quicksort': {'all': row(t, 0.2, runs, **tol)}}}})
+
+ def test_a_rebaseline_replaces_the_row(self):
+ rows, _ = self.resolve([self.rebase(12, 0.8)])
+ self.assertEqual(rows['linux-x64@a']['quicksort']['all'], row(0.8, 0.1))
+
+ def test_two_rebaselines_of_one_row_are_a_conflict_naming_both(self):
+ with self.assertRaises(baselines.BaselineError) as caught:
+ self.resolve([self.rebase(12, 0.8), self.rebase(15, 0.9)])
+ self.assertIn('pr/12.json', str(caught.exception))
+ self.assertIn('pr/15.json', str(caught.exception))
+
+ def test_a_rebaseline_from_a_stale_value_is_rejected(self):
+ # pr/12 was folded into base; pr/15 measured against the value before it.
+ moved = {'linux-x64@a': {'quicksort': {'all': row(0.8, 0.1)}}}
+ with self.assertRaises(baselines.BaselineError) as caught:
+ self.resolve([self.rebase(15, 0.9, frm=1.0)], base=moved)
+ self.assertIn('another merged change moved it first', str(caught.exception))
+
+ def test_a_rebaseline_needs_a_reason_and_a_from(self):
+ number, overlay = self.rebase(12, 0.8, reason='TODO')
+ with self.assertRaises(baselines.BaselineError):
+ baselines.validate_overlay(number, overlay, 'pr/12.json')
+ number, overlay = self.rebase(12, 0.8)
+ del overlay['rebaseline']['linux-x64@a']['quicksort']['all']['from']
+ with self.assertRaises(baselines.BaselineError):
+ baselines.validate_overlay(number, overlay, 'pr/12.json')
+
+ def test_a_rebaseline_of_a_missing_row_is_rejected(self):
+ with self.assertRaises(baselines.BaselineError):
+ self.resolve([self.rebase(12, 0.8)], base={})
+
+ def test_the_overlay_number_must_match_its_file(self):
+ with self.assertRaises(baselines.BaselineError):
+ baselines.validate_overlay(13, self.calib(12, 1.0)[1], 'pr/13.json')
+
+ def test_two_branches_calibrating_one_cpu_are_combined_in_any_order(self):
+ a, b = self.calib(12, 1.0, time=0.3), self.calib(15, 1.2, runs=2)
+ rows_ab, _ = self.resolve([a, b])
+ rows_ba, _ = self.resolve([b, a])
+ self.assertEqual(rows_ab, rows_ba)
+ combined = rows_ab['linux-x64@b']['quicksort']['all']
+ self.assertEqual((combined['time'], combined['runs']), (1.1, 3))
+ self.assertEqual(combined['tolerance'], {'time': 0.3})
+
+ def test_a_calibration_of_an_existing_row_is_superseded_not_fatal(self):
+ rows, notes = self.resolve([self.calib(12, 9.9, key='linux-x64@a')])
+ self.assertEqual(rows['linux-x64@a']['quicksort']['all']['time'], 1.0)
+ self.assertIn('superseded', notes[0])
+
+ def test_a_rebaseline_can_move_a_row_another_branch_calibrated(self):
+ frm = (15, {'pr': 15, 'reason': 'moved', 'rebaseline': {'linux-x64@b': {'quicksort': {
+ 'all': dict(row(0.5, 0.2), **{'from': {'time': 1.0, 'memory': 0.2}})}}}})
+ rows, _ = self.resolve([self.calib(12, 1.0), frm])
+ self.assertEqual(rows['linux-x64@b']['quicksort']['all']['time'], 0.5)
+
+ def test_fold_moves_everything_into_base_and_changes_no_verdict(self):
+ tree = BaselineTree(self.BASE, dict([self.rebase(12, 0.8), self.calib(15, 1.0)]))
+ try:
+ before = baselines.load(tree.root)['platforms']
+ self.assertEqual(baselines.fold(tree.root), [12, 15])
+ self.assertEqual(list((tree.root / 'pr').iterdir()), [])
+ self.assertEqual(baselines.load_base(tree.root), before)
+ self.assertEqual(sorted(p.name for p in (tree.root / 'base').iterdir()),
+ ['linux-x64@a.json', 'linux-x64@b.json'])
+ self.assertEqual(baselines.fold(tree.root), [])
+ finally:
+ tree.close()
+
+ def test_fold_refuses_contradicting_overlays(self):
+ tree = BaselineTree(self.BASE, dict([self.rebase(12, 0.8), self.rebase(15, 0.9)]))
+ try:
+ with self.assertRaises(baselines.BaselineError):
+ baselines.fold(tree.root)
+ self.assertEqual(len(list((tree.root / 'pr').iterdir())), 2)
+ finally:
+ tree.close()
+
+ def test_the_checked_in_baselines_resolve(self):
+ data = baselines.load()
+ self.assertTrue(data['platforms'])
+ for key in data['platforms']:
+ self.assertRegex(key, baselines.KEY_RE)
+
+ def test_import_legacy_takes_only_the_branchs_own_edits(self):
+ tree = BaselineTree({'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1, memory=0.3)},
+ 'recursion': {'all': row(2.0, 0.1)}}})
+ try:
+ # The branch started before master gave quicksort its RAM tolerance, so its
+ # copy has the OLD quicksort row -- that is not an edit of its own.
+ original = dict(POLICY, platforms={'linux-x64@a': {
+ 'quicksort': {'all': row(1.0, 0.1)}, 'recursion': {'all': row(2.0, 0.1)}}})
+ legacy = dict(POLICY, platforms={
+ 'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1)},
+ 'recursion': {'all': row(1.5, 0.1)}},
+ 'linux-x64@b': {'quicksort': {'all': row(2.0, 0.2, 1)}}})
+ baselines.import_legacy(tree.root, 31, legacy, original, 'imported')
+ overlay = tree.overlay(31)
+ self.assertEqual(overlay['calibrate']['linux-x64@b']['quicksort']['all']['time'], 2.0)
+ self.assertEqual(sorted(overlay['rebaseline']['linux-x64@a']), ['recursion'])
+ self.assertEqual(overlay['rebaseline']['linux-x64@a']['recursion']['all']['from'],
+ {'time': 2.0, 'memory': 0.1})
+ # A row both the branch and master calibrated stays master's, and is no rebaseline.
+ master_too = dict(legacy, platforms=dict(legacy['platforms'], **{
+ 'linux-x64@c': {'quicksort': {'all': row(3.0, 0.3, 1)}}}))
+ tree2 = BaselineTree({'linux-x64@c': {'quicksort': {'all': row(2.9, 0.3, 2)}}})
+ try:
+ path, notes = baselines.import_legacy(tree2.root, 32, master_too, original, 'x')
+ self.assertNotIn('rebaseline', tree2.overlay(32) or {})
+ self.assertTrue(any('linux-x64@c' in n for n in notes), notes)
+ finally:
+ tree2.close()
+ finally:
+ tree.close()
+
+
+class CheckTests(unittest.TestCase):
+ """perf_baseline.py check --base: what a pull request may change."""
+
+ def check(self, changed, number=31, migrating=False):
+ originals = baselines.changed_files, baselines.base_exists_at
+ baselines.changed_files = lambda base_ref, root=None: changed
+ baselines.base_exists_at = lambda ref, root=None: not migrating
+ try:
+ return baselines.check(baselines.ROOT, 'base-sha', number)
+ finally:
+ baselines.changed_files, baselines.base_exists_at = originals
+
+ def test_the_migration_that_creates_base_may(self):
+ self.assertEqual(self.check(['base/linux-x64@a.json'], migrating=True), [])
+
+ def test_a_pull_request_may_write_its_own_overlay_and_the_policy(self):
+ self.assertEqual(self.check(['pr/31.json', 'policy.json']), [])
+
+ def test_base_is_the_folds_alone(self):
+ problems = self.check(['base/linux-x64@a.json'])
+ self.assertEqual(len(problems), 1)
+ self.assertIn('nightly fold', problems[0])
+
+ def test_another_pull_requests_overlay_is_off_limits(self):
+ self.assertIn('pr/31.json', self.check(['pr/30.json'])[0])
+
+
+class SummaryTests(unittest.TestCase):
+ def test_one_entry_per_platform_with_the_spread_of_its_cpus(self):
+ rows = {'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1)}},
+ 'linux-x64@b': {'quicksort': {'all': row(1.4, 0.3)}},
+ 'linux-x64@c': {'quicksort': {'all': row(1.2, 0.2)}},
+ 'macos-arm64': {'quicksort': {'all': row(0.9, 0.1)}}}
+ s = baselines.summary(rows)
+ self.assertEqual([p['id'] for p in s['platforms']], ['linux-x64', 'macos-arm64'])
+ linux = s['platforms'][0]
+ self.assertEqual(linux['name'], 'Linux x64')
+ self.assertEqual(linux['cpus'], ['a', 'b', 'c'])
+ self.assertEqual(linux['benchmarks']['quicksort']['time'],
+ {'median': 1.2, 'min': 1.0, 'max': 1.4})
+
+ def test_every_gated_benchmark_is_described(self):
+ self.assertEqual([b for b, _, _ in baselines.BENCHMARKS],
+ ['hello', 'translator'] + gate.WORKLOADS)
+
+ def test_shared_workloads_read_as_the_absolute_table_does(self):
+ # The workloads are CommonWorkloads, which the Port Status page's absolute table
+ # also runs; one benchmark must not be described two ways on one page.
+ import json
+ support = json.loads((Path(__file__).resolve().parents[2] /
+ 'docs/website/data/port_status_support.json').read_text())
+ mine = {b: (n, d) for b, n, d in baselines.BENCHMARKS}
+ for r in support['benchmark']['rows']:
+ self.assertEqual(mine[r['id']], (r['name'], r['description']), r['id'])
class VerdictStepTests(unittest.TestCase):
@@ -248,8 +520,9 @@ def verdict(self, results):
capture_output=True, text=True)
def row(self, verdict_):
- return {'all': {'time': {'median': 1.0, 'baseline': None if verdict_ == 'uncalibrated'
- else 1.0, 'verdict': verdict_},
+ return {'all': {'time': {'median': 0.7 if verdict_ == 'improved' else 1.0,
+ 'baseline': None if verdict_ == 'uncalibrated' else 1.0,
+ 'verdict': verdict_},
'memory': {'median': 0.5, 'baseline': 0.5, 'verdict': 'ok'}}}
def test_a_missing_baseline_fails_the_job(self):
@@ -262,6 +535,13 @@ def test_a_missing_baseline_fails_the_job(self):
self.assertIn('NO BASELINE', r.stdout)
self.assertIn('calibrate-perf-baseline.py', r.stdout)
+ def test_an_improvement_fails_the_job(self):
+ r = self.verdict({'platform': 'linux-x64', 'pr': 5931, 'labels': {},
+ 'results': {'quicksort': self.row('improved')}, 'regression': False})
+ self.assertEqual(r.returncode, 1, r.stdout + r.stderr)
+ self.assertIn('IMPROVED', r.stdout)
+ self.assertIn('perf-baseline/pr/5931.json', r.stdout)
+
def test_a_judged_run_passes(self):
r = self.verdict({'platform': 'linux-x64', 'results': {'quicksort': self.row('ok')},
'regression': False})
From 0f3252a1b4eead937ac4d274c6fa7bdff85b1f57 Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 05:40:39 +0300
Subject: [PATCH 02/11] Perf baselines: repairable merged overlays, safer
combine and import, stricter check
- A pull request may edit or delete an overlay already on its base branch: two
merged rebaselines of one row otherwise leave master unresolvable with no
check-compliant repair.
- Combined calibrations get a tolerance covering every contributing row's own
band around the combined median, so none of the runs they came from fails.
- import-legacy merges a row field by field against the merge base, keeping
master's changes and refusing fields both sides changed differently.
- policy.json values are type- and range-checked; a change to the retired
perf-baseline.json fails the check with migration instructions.
- README shows the import command with the --ref flag it requires.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
vm/selfhost/README.md | 6 +-
vm/selfhost/calibrate-perf-baseline.py | 4 +-
vm/selfhost/perf_baseline.py | 144 ++++++++++++++++++++-----
vm/selfhost/test_perf_gate.py | 64 ++++++++++-
4 files changed, 182 insertions(+), 36 deletions(-)
diff --git a/vm/selfhost/README.md b/vm/selfhost/README.md
index d71c0c4cba4..13efeec3fed 100644
--- a/vm/selfhost/README.md
+++ b/vm/selfhost/README.md
@@ -140,8 +140,10 @@ No pull request edits `perf-baseline/base/`. The nightly `fold` job in
`.github/workflows/perf-baseline.yml` moves merged overlays there. The one-file layout
this replaced made unrelated branches conflict on every merge; `perf_baseline.py`
explains how overlays combine and when two of them are a real conflict. A branch still
-carrying edits to the old `perf-baseline.json` converts them with
-`perf_baseline.py import-legacy --pr N --reason "..." `.
+carrying edits to the old `perf-baseline.json` converts them, from a checkout of this
+layout, with `perf_baseline.py import-legacy --pr N --reason "..." --ref origin/`.
+It reads the branch's copy and the copy at its merge base, so only the branch's own edits
+are imported.
The Port Status page's ParparVM vs JDK 25 table is rendered from these same rows
(`perf_baseline.py summary`, run by `scripts/website/build.sh`).
diff --git a/vm/selfhost/calibrate-perf-baseline.py b/vm/selfhost/calibrate-perf-baseline.py
index 4c6fe0166e5..39064dc8cbd 100644
--- a/vm/selfhost/calibrate-perf-baseline.py
+++ b/vm/selfhost/calibrate-perf-baseline.py
@@ -120,8 +120,8 @@ def main(argv=None):
# What the gate judged against (this pull request's earlier rows included), and the
# baseline as it stands without them -- which is what a rebaseline's "from" names,
# since this run replaces this pull request's earlier rebaseline rather than stacking.
- judged, _ = perf_baseline.resolve(base, overlays)
- others, _ = perf_baseline.resolve(base, [o for o in overlays if o[0] != number])
+ judged, _ = perf_baseline.resolve(base, overlays, tolerance)
+ others, _ = perf_baseline.resolve(base, [o for o in overlays if o[0] != number], tolerance)
runs = collect(args.results, set(args.only.split(',')) if args.only else None)
widest = defaultdict(float) # (benchmark, metric) -> widest tolerance any row has
diff --git a/vm/selfhost/perf_baseline.py b/vm/selfhost/perf_baseline.py
index 9a97e756480..c313b3b689d 100644
--- a/vm/selfhost/perf_baseline.py
+++ b/vm/selfhost/perf_baseline.py
@@ -61,6 +61,7 @@
HERE = Path(__file__).resolve().parent
ROOT = HERE / 'perf-baseline'
+LEGACY_FILE = 'vm/selfhost/perf-baseline.json' # the retired single-file layout
REPO = HERE.parents[1]
METRICS = ('time', 'memory')
KEY_RE = re.compile(r'^[a-z0-9]+-[a-z0-9]+(@[a-z0-9-]+)?$')
@@ -165,8 +166,18 @@ def _rows(where, tree):
def load_policy(root=ROOT):
policy = _read_json(Path(root) / 'policy.json')
for name in ('tolerance', 'floor'):
- if set(policy.get(name, {})) != set(METRICS):
+ values = policy.get(name)
+ if not isinstance(values, dict) or set(values) != set(METRICS):
raise BaselineError('policy.json: %s must give time and memory' % name)
+ for metric, value in values.items():
+ # A pull request may edit the policy, so a value verdict() would choke on --
+ # or a negative tolerance, which inverts the gate -- is refused here, before
+ # ten minutes of measurement per platform consume it.
+ numeric = not isinstance(value, bool) and isinstance(value, (int, float))
+ if not numeric or not (value > 0 if name == 'tolerance' else value >= 0) or \
+ not value < 5:
+ raise BaselineError('policy.json: %s %s %r is not a %s fraction' % (
+ name, metric, value, 'positive' if name == 'tolerance' else 'non-negative'))
return policy
@@ -225,23 +236,40 @@ def _same(a, b):
return all(math.isclose(a[m], b[m], rel_tol=0, abs_tol=5e-4) for m in METRICS)
-def _combine(rows):
- """Several branches calibrating the same new row: one calibration from all of them."""
+def _combine(rows, default=None):
+ """Several branches calibrating the same new row: one calibration from all of them.
+
+ The baseline is the median of their ratios, and the tolerance is wide enough that every
+ contributing row's own band (its ratio, plus or minus its tolerance, or `default`'s
+ where it gives none) still passes around that median. Keeping only the widest of the
+ rows' tolerances is not enough: calibrations at 1.0x and 2.0x, each at 15%, would fold
+ to 1.5x at 15%, which neither of the runs they came from passes. The median is not
+ weighted by `runs`: each overlay's row is already the median of its own runs, and
+ two branches meeting the same new CPU in the window before a fold are rare enough
+ that a plain median of the two is the honest summary."""
if len(rows) == 1:
return copy.deepcopy(rows[0])
+ default = default or {}
combined = {m: round(statistics.median(r[m] for r in rows), 3) for m in METRICS}
combined['runs'] = sum(r['runs'] for r in rows)
tolerance = {}
- for r in rows:
- for metric, value in r.get('tolerance', {}).items():
- tolerance[metric] = max(tolerance.get(metric, 0.0), value)
+ for metric in METRICS:
+ base = combined[metric]
+ need = 0.0
+ for r in rows:
+ own = r.get('tolerance', {}).get(metric, default.get(metric, 0.0))
+ need = max(need, own, r[metric] * (1 + own) / base - 1, 1 - r[metric] * (1 - own) / base)
+ need = round(math.ceil(need / 0.05 - 1e-9) * 0.05, 2)
+ if need > default.get(metric, 0.0):
+ tolerance[metric] = need
if tolerance:
combined['tolerance'] = tolerance
return combined
-def resolve(base, overlays):
- """The rows the gate judges against: base with every overlay applied.
+def resolve(base, overlays, tolerance=None):
+ """The rows the gate judges against: base with every overlay applied. `tolerance` is
+ the policy's global one, which a row without its own is judged by.
Returns (rows, notes). Raises BaselineError for overlays that contradict each other or
the base, naming the pull requests involved."""
@@ -256,7 +284,8 @@ def resolve(base, overlays):
notes.append('%s %s/%s: already calibrated; the calibration in %s is superseded'
% (key, bench, cores, ', '.join('pr/%d.json' % n for n, _ in entries)))
continue
- rows.setdefault(key, {}).setdefault(bench, {})[cores] = _combine([r for _, r in entries])
+ rows.setdefault(key, {}).setdefault(bench, {})[cores] = _combine([r for _, r in entries],
+ tolerance)
rebaselines = {}
for number, overlay in overlays:
for key, bench, cores, row in _rows('pr/%d.json' % number, overlay.get('rebaseline', {})):
@@ -287,7 +316,7 @@ def resolve(base, overlays):
def load(root=ROOT):
"""Everything perf-gate.py needs: tolerance, floor and the resolved rows."""
policy = load_policy(root)
- rows, notes = resolve(load_base(root), load_overlays(root))
+ rows, notes = resolve(load_base(root), load_overlays(root), policy['tolerance'])
return {'tolerance': policy['tolerance'], 'floor': policy['floor'], 'platforms': rows,
'notes': notes}
@@ -308,13 +337,14 @@ def fold(root=ROOT):
folded. Verifies the gate reads the same baseline afterwards."""
root = Path(root)
overlays = load_overlays(root)
- before, _ = resolve(load_base(root), overlays)
+ tolerance = load_policy(root)['tolerance']
+ before, _ = resolve(load_base(root), overlays, tolerance)
if not overlays:
return []
write_base(root, before)
for number, _ in overlays:
(root / 'pr' / ('%d.json' % number)).unlink()
- after, _ = resolve(load_base(root), load_overlays(root))
+ after, _ = resolve(load_base(root), load_overlays(root), tolerance)
if after != before:
raise BaselineError('folding changed the resolved baseline; nothing may be committed')
return [number for number, _ in overlays]
@@ -427,12 +457,22 @@ def changed_files(base_ref, root=ROOT):
return [line[len(rel) + 1:] for line in out.stdout.splitlines() if line.strip()]
-def base_exists_at(ref, root=ROOT):
+def exists_at(ref, path, root=ROOT):
+ """Whether `path` (relative to root) exists in commit `ref`."""
rel = Path(root).resolve().relative_to(REPO).as_posix()
- return subprocess.run(['git', 'cat-file', '-e', '%s:%s/base' % (ref, rel)],
+ return subprocess.run(['git', 'cat-file', '-e', '%s:%s/%s' % (ref, rel, path)],
capture_output=True, cwd=str(REPO)).returncode == 0
+def legacy_touched(base_ref):
+ """Whether the pull request changed the retired single-file baseline."""
+ out = subprocess.run(['git', 'diff', '--name-only', '%s...HEAD' % base_ref, '--',
+ LEGACY_FILE], capture_output=True, text=True, cwd=str(REPO))
+ if out.returncode != 0:
+ raise BaselineError('git diff against %s failed: %s' % (base_ref, out.stderr.strip()))
+ return bool(out.stdout.strip())
+
+
def check(root=ROOT, base_ref=None, number=None):
"""Problems with the baselines as a list of strings; empty when the gate can use them.
With base_ref, also what a pull request may change: its own overlay and the policy,
@@ -451,21 +491,32 @@ def check(root=ROOT, base_ref=None, number=None):
return problems + [str(error)]
# The change that introduced this layout creates base/; that is the one pull
# request allowed to, and it is recognisable by base/ not existing before it.
- migrating = not base_exists_at(base_ref, root)
+ migrating = not exists_at(base_ref, 'base', root)
for path in changed:
if path.startswith('base/') and not migrating:
problems.append('%s changed: base/ is written only by the nightly fold. Put the '
'rows in pr/.json instead '
'(calibrate-perf-baseline.py --pr N writes it).' % path)
- elif path.startswith('pr/') and number is not None and path != 'pr/%d.json' % number:
+ elif path.startswith('pr/') and number is not None and path != 'pr/%d.json' % number \
+ and not exists_at(base_ref, path, root):
+ # An overlay already on the base branch belongs to a MERGED pull request, and
+ # editing or deleting it is how master is repaired: two pull requests can each
+ # pass while rebaselining the same row (neither sees the other's unmerged
+ # overlay), and once both merge, resolve() and the fold both refuse master
+ # until one of them is changed. Only another OPEN pull request's overlay --
+ # one this branch would be inventing -- is off limits.
problems.append('%s changed: a pull request writes only its own overlay, '
- 'pr/%d.json' % (path, number))
+ 'pr/%d.json, or repairs one already merged' % (path, number))
+ if not migrating and legacy_touched(base_ref):
+ # A branch from before this layout that resolves its merge conflict by keeping
+ # the old file would pass every other check, and the gate -- which reads only
+ # perf-baseline/ -- would silently ignore the rows it meant to add.
+ problems.append('%s changed: that file is retired and nothing reads it. Convert '
+ 'the branch\'s edits with `perf_baseline.py import-legacy --pr N '
+ '--ref ` and delete it.' % LEGACY_FILE)
return problems
-LEGACY_FILE = 'vm/selfhost/perf-baseline.json'
-
-
def _git_show(ref, path):
out = subprocess.run(['git', 'show', '%s:%s' % (ref, path)], capture_output=True,
text=True, cwd=str(REPO))
@@ -496,7 +547,7 @@ def import_legacy(root, number, legacy, original=None, reason=None):
rows it has become rebaselines from their current value. Returns (path or None, notes)."""
current = load(root)['platforms']
before = (original or {}).get('platforms')
- calibrate, rebaseline, notes = {}, {}, []
+ calibrate, rebaseline, notes, conflicts = {}, {}, [], []
for key, bench, cores, row in _rows('the branch perf-baseline.json', legacy.get('platforms', {})):
_check_row('%s %s/%s' % (key, bench, cores), row)
if before is not None and before.get(key, {}).get(bench, {}).get(cores) == row:
@@ -510,17 +561,56 @@ def import_legacy(root, number, legacy, original=None, reason=None):
# two calibrations of one CPU, and the one already in the tree stands.
notes.append('%s %s/%s: calibrated on master too; master\'s row stands'
% (key, bench, cores))
- elif existing != row:
- if before is not None and old is not None and old != existing:
- notes.append('%s %s/%s was also changed on master since the branch point; '
- 'check the imported value' % (key, bench, cores))
- moved = dict(row, **{'from': {m: existing[m] for m in METRICS}})
- rebaseline.setdefault(key, {}).setdefault(bench, {})[cores] = moved
+ else:
+ merged = row if old is None else _merge_fields(old, row, existing,
+ '%s %s/%s' % (key, bench, cores),
+ conflicts)
+ if merged is not None and merged != existing:
+ moved = dict(merged, **{'from': {m: existing[m] for m in METRICS}})
+ rebaseline.setdefault(key, {}).setdefault(bench, {})[cores] = moved
+ if conflicts:
+ raise BaselineError('the branch and master both changed these fields since the branch '
+ 'point, differently; re-measure them on top of master instead of '
+ 'importing:\n ' + '\n '.join(conflicts))
if not calibrate and not rebaseline:
return None, notes
return write_overlay(root, number, calibrate, rebaseline, reason), notes
+def _flat(row):
+ flat = {f: row[f] for f in ('time', 'memory', 'runs')}
+ for metric, value in row.get('tolerance', {}).items():
+ flat['tolerance.' + metric] = value
+ return flat
+
+
+def _merge_fields(old, branch, master, where, conflicts):
+ """Three-way merge of one row, field by field: the branch's value where only the
+ branch changed it, master's everywhere else. Copying the branch's whole row would
+ silently revert whatever master changed in the same row since the branch point -- a
+ tolerance master added for time, say, under a branch that only re-measured memory."""
+ o, b, m = _flat(old), _flat(branch), _flat(master)
+ merged = {}
+ for field in sorted(set(o) | set(b) | set(m)):
+ ov, bv, mv = o.get(field), b.get(field), m.get(field)
+ if bv == ov:
+ value = mv
+ elif mv in (ov, bv):
+ value = bv
+ else:
+ conflicts.append('%s %s: branch %r, master %r (was %r)' % (where, field, bv, mv, ov))
+ continue
+ if value is not None:
+ merged[field] = value
+ if any(c.startswith(where + ' ') for c in conflicts):
+ return None
+ row = {f: merged[f] for f in ('time', 'memory', 'runs')}
+ tolerance = {f.split('.', 1)[1]: v for f, v in merged.items() if f.startswith('tolerance.')}
+ if tolerance:
+ row['tolerance'] = tolerance
+ return row
+
+
def main(argv=None):
parser = argparse.ArgumentParser(description=__doc__.split('\n')[0])
parser.add_argument('--root', default=str(ROOT))
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index a2be5453e3c..ef59e1e3f7b 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -369,7 +369,20 @@ def test_two_branches_calibrating_one_cpu_are_combined_in_any_order(self):
self.assertEqual(rows_ab, rows_ba)
combined = rows_ab['linux-x64@b']['quicksort']['all']
self.assertEqual((combined['time'], combined['runs']), (1.1, 3))
- self.assertEqual(combined['tolerance'], {'time': 0.3})
+ # Wide enough for 1.0x +/- 30% around the 1.1x median: down to 0.7x is -36%.
+ self.assertEqual(combined['tolerance'], {'time': 0.4})
+
+ def test_combined_calibrations_still_pass_the_runs_they_came_from(self):
+ tol = POLICY['tolerance']
+ a, b = self.calib(12, 1.0), self.calib(15, 2.0)
+ for n, o in (a, b):
+ baselines.validate_overlay(n, o, 'pr/%d.json' % n)
+ rows, _ = baselines.resolve(self.BASE, [a, b], tol)
+ r = rows['linux-x64@b']['quicksort']['all']
+ self.assertEqual(r['time'], 1.5)
+ for ratio in (1.0, 2.0, 1.0 * 0.85, 2.0 * 1.15):
+ self.assertEqual(gate.verdict(ratio, r['time'], r['tolerance']['time']), 'ok', ratio)
+ self.assertNotIn('memory', r.get('tolerance', {})) # 0.2 and 0.2: the global 15%
def test_a_calibration_of_an_existing_row_is_superseded_not_fatal(self):
rows, notes = self.resolve([self.calib(12, 9.9, key='linux-x64@a')])
@@ -410,6 +423,35 @@ def test_the_checked_in_baselines_resolve(self):
for key in data['platforms']:
self.assertRegex(key, baselines.KEY_RE)
+ def test_import_legacy_keeps_what_master_changed_in_the_same_row(self):
+ # master added a time tolerance after the branch point; the branch re-measured RAM.
+ tree = BaselineTree({'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1, time=0.4)}}})
+ try:
+ original = dict(POLICY, platforms={'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1)}}})
+ legacy = dict(POLICY, platforms={'linux-x64@a': {'quicksort': {'all': row(1.0, 0.3)}}})
+ baselines.import_legacy(tree.root, 33, legacy, original, 'ram re-measured')
+ r = tree.overlay(33)['rebaseline']['linux-x64@a']['quicksort']['all']
+ self.assertEqual((r['memory'], r['tolerance']), (0.3, {'time': 0.4}))
+ # Both sides changing the SAME field differently is refused, not guessed.
+ clash = dict(POLICY, platforms={'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1, time=0.2)}}})
+ with self.assertRaises(baselines.BaselineError) as caught:
+ baselines.import_legacy(tree.root, 34, clash, original, 'x')
+ self.assertIn('tolerance.time', str(caught.exception))
+ finally:
+ tree.close()
+
+ def test_a_broken_policy_is_refused(self):
+ import json
+ for bad in ({'time': '0.15', 'memory': 0.15}, {'time': -0.1, 'memory': 0.15},
+ {'time': 0, 'memory': 0.15}):
+ tree = BaselineTree()
+ try:
+ (tree.root / 'policy.json').write_text(json.dumps(dict(POLICY, tolerance=bad)))
+ with self.assertRaises(baselines.BaselineError):
+ baselines.load_policy(tree.root)
+ finally:
+ tree.close()
+
def test_import_legacy_takes_only_the_branchs_own_edits(self):
tree = BaselineTree({'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1, memory=0.3)},
'recursion': {'all': row(2.0, 0.1)}}})
@@ -445,14 +487,26 @@ def test_import_legacy_takes_only_the_branchs_own_edits(self):
class CheckTests(unittest.TestCase):
"""perf_baseline.py check --base: what a pull request may change."""
- def check(self, changed, number=31, migrating=False):
- originals = baselines.changed_files, baselines.base_exists_at
+ def check(self, changed, number=31, migrating=False, merged=(), legacy=False):
+ originals = baselines.changed_files, baselines.exists_at, baselines.legacy_touched
baselines.changed_files = lambda base_ref, root=None: changed
- baselines.base_exists_at = lambda ref, root=None: not migrating
+ baselines.exists_at = lambda ref, path, root=None: (
+ not migrating if path == 'base' else path in merged)
+ baselines.legacy_touched = lambda base_ref: legacy
try:
return baselines.check(baselines.ROOT, 'base-sha', number)
finally:
- baselines.changed_files, baselines.base_exists_at = originals
+ baselines.changed_files, baselines.exists_at, baselines.legacy_touched = originals
+
+ def test_a_merged_overlay_may_be_repaired(self):
+ # Two merged pull requests that rebaselined one row leave master unresolvable;
+ # editing or deleting one of their overlays is the fix, and must pass.
+ self.assertEqual(self.check(['pr/30.json'], merged=('pr/30.json',)), [])
+
+ def test_the_retired_file_is_refused(self):
+ problems = self.check([], legacy=True)
+ self.assertEqual(len(problems), 1)
+ self.assertIn('import-legacy', problems[0])
def test_the_migration_that_creates_base_may(self):
self.assertEqual(self.check(['base/linux-x64@a.json'], migrating=True), [])
From b2d883fa6a05828f606584353227b1be6f510f25 Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 07:21:53 +0300
Subject: [PATCH 03/11] Perf baselines: recalibrate bimodal stringBuilding RAM;
tolerance-aware stale check
CI on this branch failed macOS and Windows x64 (AMD Family 25 Model 1) as
IMPROVED on stringBuilding RAM. Not an improvement: across 79 CI runs that row
reads 0.267x or 0.322x on that Windows CPU, and 0.389x-0.54x on macOS, with
unchanged VM code, and the baselines were calibrated from runs on the upper
mode. pr/5930.json recalibrates the RAM of those two rows from all their runs;
replaying the 30 runs whose VM and benchmark sources equal master's now gives
0 non-ok verdicts of 780.
- calibrate-perf-baseline.py feeds a run to the row that JUDGED it (macOS reports
'Apple M1 (Virtual)' but is judged by the plain macos-arm64 row), and gains
--metric to recalibrate one metric without touching the other.
- A rebaseline's 'from' records the replaced row's tolerance too, so a later
rebaseline cannot silently restore a tolerance an earlier one changed.
- import-legacy refuses rows the branch deleted (the old --fresh), which an
overlay cannot express, instead of reporting no change.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
vm/selfhost/calibrate-perf-baseline.py | 28 ++++++++++----
vm/selfhost/perf-baseline/pr/5930.json | 42 ++++++++++++++++++++
vm/selfhost/perf_baseline.py | 47 ++++++++++++++++++-----
vm/selfhost/test_perf_gate.py | 53 ++++++++++++++++++++++++--
4 files changed, 151 insertions(+), 19 deletions(-)
create mode 100644 vm/selfhost/perf-baseline/pr/5930.json
diff --git a/vm/selfhost/calibrate-perf-baseline.py b/vm/selfhost/calibrate-perf-baseline.py
index 39064dc8cbd..b4aa2576178 100644
--- a/vm/selfhost/calibrate-perf-baseline.py
+++ b/vm/selfhost/calibrate-perf-baseline.py
@@ -2,7 +2,8 @@
"""Writes a pull request's baseline changes, pr/.json, from CI runs' perf-results.json.
python3 vm/selfhost/calibrate-perf-baseline.py [--pr N] [--reason TEXT] [--all]
- [--only BENCH,...] RESULTS.json ...
+ [--only BENCH,...] [--metric time|memory]
+ RESULTS.json ...
Pass the perf-results.json of the runs the gate failed on -- from the artifacts
ci-perf-gate.sh leaves (linux-screenshot-raw-*/perf, windows-port-screenshot-raw-*/perf,
@@ -17,7 +18,8 @@
moved performance on purpose. Only the metric that moved is replaced, and
--reason is required -- the reason is what a reviewer reads beside the
number. --all rebaselines every measured row instead, for recalibrating a
- row's noise from several runs of unchanged code.
+ row's noise from several runs of unchanged code; --metric limits that to one
+ metric, so recalibrating a noisy RAM figure leaves a steady time alone.
How a row is computed:
baseline the median of the runs' median ratios.
@@ -79,15 +81,23 @@ def spread_tolerance(values, base, floor):
return max(floor, round_up(spread * SPREAD_MARGIN))
-def collect(paths, only=None):
- """platform-key -> (benchmark, cores) -> metric -> [median ratio per run]."""
+def collect(paths, only=None, rows=None):
+ """platform-key -> (benchmark, cores) -> metric -> [median ratio per run].
+
+ A run feeds the row that JUDGED it, chosen exactly as the gate chooses
+ (perf_gate.baseline_key over `rows`): macOS reports "Apple M1 (Virtual)" yet is judged
+ by the plain macos-arm64 row, and keying it by model would have rebaselined nothing and
+ calibrated a new per-model row beside the one that failed. A run no row judged
+ calibrates its own CPU model's row."""
runs = defaultdict(lambda: defaultdict(lambda: defaultdict(list)))
for path in paths:
report = json.loads(Path(path).read_text())
if report.get('error') or report.get('failures'):
raise SystemExit('%s did not complete; calibrate only from complete runs' % path)
cls = perf_gate.cpu_class(report.get('cpu'))
- key = '%s@%s' % (report['platform'], cls) if cls else report['platform']
+ key = perf_gate.baseline_key(rows or {}, report['platform'], report.get('cpu'))
+ if key is None or key not in (rows or {}):
+ key = '%s@%s' % (report['platform'], cls) if cls else report['platform']
for bench, by_cores in report['results'].items():
if only and bench not in only:
continue
@@ -105,6 +115,8 @@ def main(argv=None):
parser.add_argument('--all', action='store_true',
help='rebaseline every measured row, not only those out of tolerance')
parser.add_argument('--only', help='comma-separated benchmark ids to take from the runs')
+ parser.add_argument('--metric', choices=METRICS,
+ help='rebaseline only this metric (a new row always gets both)')
parser.add_argument('results', nargs='+')
args = parser.parse_args(argv)
@@ -122,7 +134,7 @@ def main(argv=None):
# since this run replaces this pull request's earlier rebaseline rather than stacking.
judged, _ = perf_baseline.resolve(base, overlays, tolerance)
others, _ = perf_baseline.resolve(base, [o for o in overlays if o[0] != number], tolerance)
- runs = collect(args.results, set(args.only.split(',')) if args.only else None)
+ runs = collect(args.results, set(args.only.split(',')) if args.only else None, judged)
widest = defaultdict(float) # (benchmark, metric) -> widest tolerance any row has
for _, bench, _, row in perf_baseline._rows('baseline', judged):
@@ -145,6 +157,8 @@ def main(argv=None):
moved = []
for metric in METRICS:
values = metrics[metric]
+ if args.metric and metric != args.metric and not new_row:
+ continue
if not (args.all or new_row or any(
perf_gate.verdict(v, current[metric],
current.get('tolerance', {}).get(metric, tolerance[metric]),
@@ -168,7 +182,7 @@ def main(argv=None):
if new_row or before is None:
calibrate.setdefault(key, {}).setdefault(bench, {})[cores] = row
else:
- row['from'] = {m: before[m] for m in METRICS}
+ row['from'] = perf_baseline.from_row(before)
rebaseline.setdefault(key, {}).setdefault(bench, {})[cores] = row
if not calibrate and not rebaseline:
diff --git a/vm/selfhost/perf-baseline/pr/5930.json b/vm/selfhost/perf-baseline/pr/5930.json
new file mode 100644
index 00000000000..25c4ddff80d
--- /dev/null
+++ b/vm/selfhost/perf-baseline/pr/5930.json
@@ -0,0 +1,42 @@
+{
+ "pr": 5930,
+ "reason": "stringBuilding RAM is bimodal on these runners with unchanged VM code: Windows x64 AMD Family 25 Model 1 reads 0.267x or 0.322x and macOS arm64 0.389x-0.54x across runs (CI runs 36824868723..36957766166). The rows were calibrated from runs that all landed on the upper mode, so the lower one failed as an improvement once improvements fail the gate. Recalibrated from all of each runner's runs; time is untouched.",
+ "rebaseline": {
+ "macos-arm64": {
+ "stringBuilding": {
+ "all": {
+ "from": {
+ "memory": 0.54,
+ "time": 0.842,
+ "tolerance": {
+ "time": 0.2
+ }
+ },
+ "memory": 0.463,
+ "runs": 5,
+ "time": 0.842,
+ "tolerance": {
+ "memory": 0.25,
+ "time": 0.2
+ }
+ }
+ }
+ },
+ "windows-x64@amd64-family-25-model-1-authenticamd": {
+ "stringBuilding": {
+ "all": {
+ "from": {
+ "memory": 0.322,
+ "time": 1.216
+ },
+ "memory": 0.322,
+ "runs": 10,
+ "time": 1.216,
+ "tolerance": {
+ "memory": 0.3
+ }
+ }
+ }
+ }
+ }
+}
diff --git a/vm/selfhost/perf_baseline.py b/vm/selfhost/perf_baseline.py
index c313b3b689d..ef446e9cea9 100644
--- a/vm/selfhost/perf_baseline.py
+++ b/vm/selfhost/perf_baseline.py
@@ -220,9 +220,10 @@ def validate_overlay(number, overlay, where):
here = '%s rebaseline %s %s/%s' % (where, key, bench, cores)
_check_row(here, row, extra=('from',))
old = row.get('from')
- if not isinstance(old, dict) or set(old) != set(METRICS):
+ if not isinstance(old, dict) or set(old) - {'tolerance'} != set(METRICS) or \
+ not isinstance(old.get('tolerance', {}), dict):
raise BaselineError('%s: "from" must give the time and memory baseline it '
- 'replaces' % here)
+ 'replaces, and its tolerance if it had one' % here)
if rebaselines:
reason = overlay.get('reason')
if not isinstance(reason, str) or not reason.strip() or reason.strip().upper().startswith('TODO'):
@@ -232,8 +233,22 @@ def validate_overlay(number, overlay, where):
raise BaselineError('%s changes nothing; delete it' % where)
+def from_row(row):
+ """What a rebaseline records as the row it replaces: the ratios AND the tolerance. A
+ tolerance-only recalibration (an --all run whose ratios round to the old values) is a
+ change too, and a later rebaseline that compared ratios alone would pass and then put
+ the stale tolerance back."""
+ old = {m: row[m] for m in METRICS}
+ if row.get('tolerance'):
+ old['tolerance'] = dict(row['tolerance'])
+ return old
+
+
def _same(a, b):
- return all(math.isclose(a[m], b[m], rel_tol=0, abs_tol=5e-4) for m in METRICS)
+ """Whether the row a rebaseline was measured against (`a`, its "from") is still the
+ row in the tree (`b`). No tolerance on either side means none."""
+ return all(math.isclose(a[m], b[m], rel_tol=0, abs_tol=5e-4) for m in METRICS) and \
+ a.get('tolerance', {}) == b.get('tolerance', {})
def _combine(rows, default=None):
@@ -304,11 +319,10 @@ def resolve(base, overlays, tolerance=None):
'"calibrate" entry' % (number, where))
if not _same(row['from'], current):
raise BaselineError(
- 'pr/%d.json rebaselines %s from time %.3fx / RAM %.3fx, but the row is now '
- 'time %.3fx / RAM %.3fx: another merged change moved it first. Re-measure '
- 'on top of it and update the rebaseline.'
- % (number, where, row['from']['time'], row['from']['memory'],
- current['time'], current['memory']))
+ 'pr/%d.json rebaselines %s from %s, but the row is now %s: another merged '
+ 'change moved it first. Re-measure on top of it and update the rebaseline.'
+ % (number, where, json.dumps(row['from'], sort_keys=True),
+ json.dumps(from_row(current), sort_keys=True)))
rows[key][bench][cores] = {k: v for k, v in row.items() if k != 'from'}
return rows, notes
@@ -548,6 +562,21 @@ def import_legacy(root, number, legacy, original=None, reason=None):
current = load(root)['platforms']
before = (original or {}).get('platforms')
calibrate, rebaseline, notes, conflicts = {}, {}, [], []
+ if before is not None:
+ # The old calibrator's --fresh dropped every row a run did not re-measure, so a
+ # branch's file can DELETE rows. An overlay has no way to say that, and importing
+ # only the rows still present would report "no change" while the rows the branch
+ # meant to retire stay active. Refuse, and say what to do instead.
+ after = legacy.get('platforms', {})
+ dropped = ['%s %s/%s' % (key, bench, cores) for key, bench, cores, _ in _rows(
+ 'the original perf-baseline.json', before)
+ if cores not in after.get(key, {}).get(bench, {})]
+ if dropped:
+ raise BaselineError(
+ 'the branch deleted %d row(s) that overlays cannot delete: %s. Re-measure '
+ 'them with calibrate-perf-baseline.py --all instead, or ask for a removal '
+ 'from base/ in its own change.' % (len(dropped), ', '.join(dropped[:8]) +
+ (' ...' if len(dropped) > 8 else '')))
for key, bench, cores, row in _rows('the branch perf-baseline.json', legacy.get('platforms', {})):
_check_row('%s %s/%s' % (key, bench, cores), row)
if before is not None and before.get(key, {}).get(bench, {}).get(cores) == row:
@@ -566,7 +595,7 @@ def import_legacy(root, number, legacy, original=None, reason=None):
'%s %s/%s' % (key, bench, cores),
conflicts)
if merged is not None and merged != existing:
- moved = dict(merged, **{'from': {m: existing[m] for m in METRICS}})
+ moved = dict(merged, **{'from': from_row(existing)})
rebaseline.setdefault(key, {}).setdefault(bench, {})[cores] = moved
if conflicts:
raise BaselineError('the branch and master both changed these fields since the branch '
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index ef59e1e3f7b..fe714da4755 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -179,7 +179,7 @@ class CalibrationTest(unittest.TestCase):
into the pull request's own overlay."""
def run_calibration(self, runs, existing=None, overlays=None, pr=7, reason=None,
- everything=False):
+ everything=False, metric=None):
import json
tree = BaselineTree(existing, overlays)
try:
@@ -196,7 +196,8 @@ def run_calibration(self, runs, existing=None, overlays=None, pr=7, reason=None,
files.append(str(f))
calibrate.main(['--root', str(tree.root), '--pr', str(pr)] +
(['--reason', reason] if reason else []) +
- (['--all'] if everything else []) + files)
+ (['--all'] if everything else []) +
+ (['--metric', metric] if metric else []) + files)
return tree.overlay(pr), baselines.load(tree.root)['platforms']
finally:
tree.close()
@@ -260,6 +261,16 @@ def test_a_run_with_no_known_cpu_feeds_the_plain_row(self):
self.assertEqual(sorted(b), ['macos-arm64'])
self.assertEqual(gate.baseline_key(b, 'macos-arm64', 'Apple M1 (Virtual)'), 'macos-arm64')
+ def test_a_run_feeds_the_row_that_judged_it(self):
+ # macOS reports a CPU model but is judged by the plain platform row.
+ old = {'macos-arm64': {'quicksort': {'all': row(1.0, 0.1)}}}
+ overlay, rows = self.run_calibration(
+ [('macos-arm64', {'quicksort': (0.5, 0.1)}, 'Apple M1 (Virtual)')],
+ existing=old, reason='faster')
+ self.assertEqual(sorted(overlay['rebaseline']), ['macos-arm64'])
+ self.assertNotIn('calibrate', overlay)
+ self.assertEqual(sorted(rows), ['macos-arm64'])
+
def test_a_row_inside_its_tolerance_is_left_alone(self):
old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1)},
'recursion': {'all': row(1.0, 0.1)}}}
@@ -276,7 +287,7 @@ def test_a_moved_row_is_rebaselined_from_its_old_value_and_needs_a_reason(self):
overlay, rows = self.run_calibration([('linux-x64', {'quicksort': (0.7, 0.1)})],
existing=old, reason='faster partitioning')
r = overlay['rebaseline']['linux-x64']['quicksort']['all']
- self.assertEqual(r['from'], {'time': 1.0, 'memory': 0.1})
+ self.assertEqual(r['from'], {'time': 1.0, 'memory': 0.1, 'tolerance': {'memory': 0.3}})
self.assertEqual(r['time'], 0.7)
# Only the metric that moved is replaced; RAM keeps its baseline and tolerance.
self.assertEqual(r['memory'], 0.1)
@@ -292,6 +303,17 @@ def test_all_recalibrates_every_measured_row(self):
r = overlay['rebaseline']['linux-x64']['quicksort']['all']
self.assertEqual((r['time'], r['runs']), (1.0, 5))
+ def test_metric_limits_a_recalibration_to_one_metric(self):
+ old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.3, time=0.4)}}}
+ overlay, _ = self.run_calibration(
+ [('linux-x64', {'quicksort': (t, m)}) for t, m in
+ ((1.1, 0.32), (0.9, 0.27), (1.05, 0.32), (0.95, 0.27), (1.0, 0.32))],
+ existing=old, reason='bimodal RAM', everything=True, metric='memory')
+ r = overlay['rebaseline']['linux-x64']['quicksort']['all']
+ self.assertEqual((r['time'], r['tolerance']['time']), (1.0, 0.4)) # untouched
+ self.assertEqual(r['memory'], 0.32)
+ self.assertGreaterEqual(r['tolerance']['memory'], 0.25) # covers the 0.27 mode
+
def test_rerunning_replaces_this_pull_requests_own_rows(self):
old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1)}}}
mine = {'pr': 7, 'reason': 'first try', 'rebaseline': {'linux-x64': {'quicksort': {
@@ -345,6 +367,16 @@ def test_a_rebaseline_from_a_stale_value_is_rejected(self):
self.resolve([self.rebase(15, 0.9, frm=1.0)], base=moved)
self.assertIn('another merged change moved it first', str(caught.exception))
+ def test_a_rebaseline_from_a_row_whose_tolerance_moved_is_rejected(self):
+ # An earlier --all recalibration changed only the tolerance; the ratios match.
+ retuned = {'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1, time=0.4)}}}
+ with self.assertRaises(baselines.BaselineError):
+ self.resolve([self.rebase(15, 0.8)], base=retuned)
+ number, overlay = self.rebase(15, 0.8)
+ overlay['rebaseline']['linux-x64@a']['quicksort']['all']['from']['tolerance'] = {'time': 0.4}
+ rows, _ = self.resolve([(number, overlay)], base=retuned)
+ self.assertEqual(rows['linux-x64@a']['quicksort']['all']['time'], 0.8)
+
def test_a_rebaseline_needs_a_reason_and_a_from(self):
number, overlay = self.rebase(12, 0.8, reason='TODO')
with self.assertRaises(baselines.BaselineError):
@@ -440,6 +472,21 @@ def test_import_legacy_keeps_what_master_changed_in_the_same_row(self):
finally:
tree.close()
+ def test_import_legacy_refuses_deleted_rows(self):
+ # The old calibrator's --fresh dropped rows a run did not re-measure.
+ tree = BaselineTree({'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1)},
+ 'recursion': {'all': row(2.0, 0.1)}}})
+ try:
+ original = dict(POLICY, platforms={'linux-x64@a': {
+ 'quicksort': {'all': row(1.0, 0.1)}, 'recursion': {'all': row(2.0, 0.1)}}})
+ legacy = dict(POLICY, platforms={'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1)}}})
+ with self.assertRaises(baselines.BaselineError) as caught:
+ baselines.import_legacy(tree.root, 35, legacy, original, 'x')
+ self.assertIn('linux-x64@a recursion/all', str(caught.exception))
+ self.assertIsNone(tree.overlay(35))
+ finally:
+ tree.close()
+
def test_a_broken_policy_is_refused(self):
import json
for bad in ({'time': '0.15', 'memory': 0.15}, {'time': -0.1, 'memory': 0.15},
From f6138aaeb7a89696399712790a23c2c622bbaf4e Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 08:36:59 +0300
Subject: [PATCH 04/11] Perf baselines: recalibrate Zen 3 translator and hello
time; refuse legacy policy edits
CI failed Windows x64 (AMD Family 25 Model 1) as IMPROVED on translator time,
0.48x against 0.61x, on this branch's unchanged VM code. The row is bimodal:
rounds alternate 0.42x-0.49x and 0.59x-0.64x inside single runs, so a run's median
lands on either mode (this branch's own runs read 0.594x and 0.479x). hello time on
the same CPU is as wide (rounds 0.87x-1.57x) and had already fallen outside on a
master run. pr/5930.json recalibrates the time of both rows from all 11 of that
CPU's runs, leaving RAM alone; the 32 runs whose VM and benchmark sources equal
master's now give 0 non-ok verdicts of 832.
import-legacy now refuses a branch that changed the old file's tolerance or floor,
which an overlay cannot carry, instead of reporting no change and losing the edit.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
vm/selfhost/perf-baseline/pr/5930.json | 30 +++++++++++++++++++++++++-
vm/selfhost/perf_baseline.py | 12 +++++++++++
vm/selfhost/test_perf_gate.py | 11 ++++++++++
3 files changed, 52 insertions(+), 1 deletion(-)
diff --git a/vm/selfhost/perf-baseline/pr/5930.json b/vm/selfhost/perf-baseline/pr/5930.json
index 25c4ddff80d..83e61d1964b 100644
--- a/vm/selfhost/perf-baseline/pr/5930.json
+++ b/vm/selfhost/perf-baseline/pr/5930.json
@@ -1,6 +1,6 @@
{
"pr": 5930,
- "reason": "stringBuilding RAM is bimodal on these runners with unchanged VM code: Windows x64 AMD Family 25 Model 1 reads 0.267x or 0.322x and macOS arm64 0.389x-0.54x across runs (CI runs 36824868723..36957766166). The rows were calibrated from runs that all landed on the upper mode, so the lower one failed as an improvement once improvements fail the gate. Recalibrated from all of each runner's runs; time is untouched.",
+ "reason": "Rows whose readings are bimodal or wide on these runners with unchanged VM code, calibrated from runs that all landed on one side, so the other side failed once improvements fail the gate. Recalibrated from all of each runner's CI runs, one metric each, the other left as it was: stringBuilding RAM on Windows x64 AMD Family 25 Model 1 (0.267x or 0.322x, 10 runs) and on macOS arm64 (0.389x-0.54x, 5 runs); translator time on that Windows CPU (rounds alternate 0.42x-0.49x and 0.59x-0.64x inside single runs; run medians 0.467x-0.628x, 11 runs) and hello time there (rounds 0.87x-1.57x, run medians 1.051x-1.352x, 11 runs).",
"rebaseline": {
"macos-arm64": {
"stringBuilding": {
@@ -23,6 +23,20 @@
}
},
"windows-x64@amd64-family-25-model-1-authenticamd": {
+ "hello": {
+ "all": {
+ "from": {
+ "memory": 0.803,
+ "time": 1.241
+ },
+ "memory": 0.803,
+ "runs": 11,
+ "time": 1.228,
+ "tolerance": {
+ "time": 0.25
+ }
+ }
+ },
"stringBuilding": {
"all": {
"from": {
@@ -36,6 +50,20 @@
"memory": 0.3
}
}
+ },
+ "translator": {
+ "all": {
+ "from": {
+ "memory": 0.542,
+ "time": 0.606
+ },
+ "memory": 0.542,
+ "runs": 11,
+ "time": 0.605,
+ "tolerance": {
+ "time": 0.35
+ }
+ }
}
}
}
diff --git a/vm/selfhost/perf_baseline.py b/vm/selfhost/perf_baseline.py
index ef446e9cea9..16855bc6526 100644
--- a/vm/selfhost/perf_baseline.py
+++ b/vm/selfhost/perf_baseline.py
@@ -562,6 +562,18 @@ def import_legacy(root, number, legacy, original=None, reason=None):
current = load(root)['platforms']
before = (original or {}).get('platforms')
calibrate, rebaseline, notes, conflicts = {}, {}, [], []
+ # The old file also carried the policy (tolerance, floor). An overlay holds rows only, so
+ # a branch that changed the policy would otherwise import as "no change" and lose it once
+ # check() rejects the retired file. policy.json is one small file a pull request may edit
+ # directly, so the edit is refused here with that instruction rather than migrated.
+ if original is not None:
+ moved_policy = [name for name in ('tolerance', 'floor')
+ if legacy.get(name) != original.get(name)]
+ if moved_policy:
+ raise BaselineError(
+ 'the branch changed the gate policy (%s) in the old file; overlays carry rows '
+ 'only. Make the same change in vm/selfhost/perf-baseline/policy.json in the '
+ 'pull request, then import the rows again.' % ', '.join(moved_policy))
if before is not None:
# The old calibrator's --fresh dropped every row a run did not re-measure, so a
# branch's file can DELETE rows. An overlay has no way to say that, and importing
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index fe714da4755..2c49a835702 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -487,6 +487,17 @@ def test_import_legacy_refuses_deleted_rows(self):
finally:
tree.close()
+ def test_import_legacy_refuses_policy_edits(self):
+ tree = BaselineTree(self.BASE)
+ try:
+ original = dict(POLICY, platforms={})
+ legacy = dict(POLICY, tolerance={'time': 0.2, 'memory': 0.15}, platforms={})
+ with self.assertRaises(baselines.BaselineError) as caught:
+ baselines.import_legacy(tree.root, 36, legacy, original, 'x')
+ self.assertIn('policy.json', str(caught.exception))
+ finally:
+ tree.close()
+
def test_a_broken_policy_is_refused(self):
import json
for bad in ({'time': '0.15', 'memory': 0.15}, {'time': -0.1, 'memory': 0.15},
From 7fe248f5c78915a6aaa2f9c68372860f042e0c1f Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 09:17:47 +0300
Subject: [PATCH 05/11] Perf baselines: a stale own overlay is re-measured, not
a dead end; stricter values
A pull request whose own rebaseline goes stale (another merged change moved the row)
was told to re-measure on top of that change, but the gate then refused to measure
and the calibrator refused to load, so no run existed to re-measure from. The gate
now judges such a run against the baseline without the pull request's own overlay
and fails afterwards, naming it; the calibrator does the same and rewrites every
stale own row from the fresh runs, refusing if the runs do not cover one.
Also: ratios and tolerances must be finite (JSON's 1e999 is infinity), a
rebaseline's 'from' values are validated like the row, and calibrations of one CPU
that combine into an out-of-range tolerance are refused, naming the overlays,
instead of resolving to a row that gates nothing.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
vm/selfhost/calibrate-perf-baseline.py | 22 +++++++-
vm/selfhost/ci-perf-gate.sh | 7 +++
vm/selfhost/perf-gate.py | 43 ++++++++++----
vm/selfhost/perf_baseline.py | 48 ++++++++++------
vm/selfhost/test_perf_gate.py | 78 ++++++++++++++++++++++++++
5 files changed, 170 insertions(+), 28 deletions(-)
diff --git a/vm/selfhost/calibrate-perf-baseline.py b/vm/selfhost/calibrate-perf-baseline.py
index b4aa2576178..6174cce4f9b 100644
--- a/vm/selfhost/calibrate-perf-baseline.py
+++ b/vm/selfhost/calibrate-perf-baseline.py
@@ -132,8 +132,20 @@ def main(argv=None):
# What the gate judged against (this pull request's earlier rows included), and the
# baseline as it stands without them -- which is what a rebaseline's "from" names,
# since this run replaces this pull request's earlier rebaseline rather than stacking.
- judged, _ = perf_baseline.resolve(base, overlays, tolerance)
others, _ = perf_baseline.resolve(base, [o for o in overlays if o[0] != number], tolerance)
+ try:
+ judged, _ = perf_baseline.resolve(base, overlays, tolerance)
+ except perf_baseline.BaselineError:
+ # This pull request's own overlay went stale (another merged change moved a row it
+ # rebaselines), and perf-gate.py then judged the run without it. Do the same, and
+ # rewrite every stale row below from these fresh runs -- which is the recovery the
+ # stale-overlay message asks for.
+ judged = others
+ stale = set()
+ for key, bench, cores, row in perf_baseline._rows('own', own.get('rebaseline', {})):
+ now = others.get(key, {}).get(bench, {}).get(cores)
+ if now is None or not perf_baseline._same(row['from'], now):
+ stale.add((key, bench, cores))
runs = collect(args.results, set(args.only.split(',')) if args.only else None, judged)
widest = defaultdict(float) # (benchmark, metric) -> widest tolerance any row has
@@ -159,7 +171,7 @@ def main(argv=None):
values = metrics[metric]
if args.metric and metric != args.metric and not new_row:
continue
- if not (args.all or new_row or any(
+ if not (args.all or new_row or (key, bench, cores) in stale or any(
perf_gate.verdict(v, current[metric],
current.get('tolerance', {}).get(metric, tolerance[metric]),
floor[metric]) != 'ok' for v in values)):
@@ -185,6 +197,12 @@ def main(argv=None):
row['from'] = perf_baseline.from_row(before)
rebaseline.setdefault(key, {}).setdefault(bench, {})[cores] = row
+ left = sorted(stale - {(k, b, c) for k, benches in rebaseline.items()
+ for b, per in benches.items() for c in per})
+ if left:
+ raise SystemExit('These rows of pr/%d.json are stale and these runs did not measure '
+ 'them; pass runs that do: %s' % (number, ', '.join(
+ '%s %s/%s' % row for row in left)))
if not calibrate and not rebaseline:
print('Every measured row is inside its tolerance; nothing to write.')
return 0
diff --git a/vm/selfhost/ci-perf-gate.sh b/vm/selfhost/ci-perf-gate.sh
index fd61cace401..d4cc55ebdab 100755
--- a/vm/selfhost/ci-perf-gate.sh
+++ b/vm/selfhost/ci-perf-gate.sh
@@ -84,6 +84,13 @@ if improved:
print('Record the new baseline in %s:' % overlay)
print(rebaseline)
sys.exit(1)
+if report.get('stale_overlay'):
+ # Measured against the baseline without this pull request's overlay, because another
+ # merged change moved a row it rebaselines; the overlay has to be re-measured.
+ print('ParparVM performance gate: %s is stale: %s' % (overlay, report['stale_overlay']))
+ print('Re-measure it from this job\'s perf-results.json:')
+ print(rebaseline)
+ sys.exit(1)
print('ParparVM performance gate: no regression on %s' % report['platform'])
PY
fi
diff --git a/vm/selfhost/perf-gate.py b/vm/selfhost/perf-gate.py
index fb95a19b3a0..5cfd137b6c1 100644
--- a/vm/selfhost/perf-gate.py
+++ b/vm/selfhost/perf-gate.py
@@ -574,13 +574,19 @@ def render_markdown(report):
'`%s`, with the reason, from this job\'s `perf-results.json`. The rebaseline '
'is then part of this pull request\'s diff:' % overlay_name(report), '',
'```', fix_command(report, True), '```']
+ if report.get('stale_overlay'):
+ lines += ['', '**`%s` is stale, so this gate fails.** %s This run was judged against '
+ 'the baseline without it; re-measure from this job\'s `perf-results.json`, '
+ 'which rewrites the stale rows:' % (overlay_name(report), report['stale_overlay']),
+ '', '```', fix_command(report, True), '```']
lines += ['', '**Result: %s**' % (
'performance regression' if report['regression'] else
('gate did not complete' if report.get('error') else
('benchmark failed' if report.get('failures') else
('no baseline: calibration required' if missing_baselines(report)
else ('improved past the baseline: rebaseline required' if improvements
- else 'no regression')))))]
+ else ('stale overlay: re-measure required' if report.get('stale_overlay')
+ else 'no regression'))))))]
return '\n'.join(lines) + '\n'
@@ -634,13 +640,28 @@ def write():
try:
baseline = perf_baseline.load(args.baseline)
except perf_baseline.BaselineError as error:
- # Overlays that contradict each other leave nothing to judge against. Say which,
- # in the comment, rather than measuring for ten minutes first.
- report['tolerance'] = {'time': 0.0, 'memory': 0.0}
- report['error'] = 'perf-baseline: %s' % error
- write()
- print('perf-gate: REFUSING: %s' % report['error'], flush=True)
- return 2
+ baseline = None
+ own = report['pr'] and (Path(args.baseline) / 'pr' / ('%d.json' % report['pr'])).is_file()
+ if own:
+ # This pull request's OWN overlay went stale: another merged change moved a row
+ # it rebaselines. The fix is to re-measure on top of that change, which needs a
+ # measurement -- so measure against the baseline without the stale overlay and
+ # fail afterwards, rather than refuse and leave nothing to recalibrate from.
+ try:
+ baseline = perf_baseline.load(args.baseline, exclude=report['pr'])
+ report['stale_overlay'] = str(error)
+ print('perf-gate: this pull request\'s overlay is stale; measuring against '
+ 'the baseline without it: %s' % error, flush=True)
+ except perf_baseline.BaselineError:
+ baseline = None
+ if baseline is None:
+ # Overlays that contradict each other leave nothing to judge against. Say which,
+ # in the comment, rather than measuring for ten minutes first.
+ report['tolerance'] = {'time': 0.0, 'memory': 0.0}
+ report['error'] = 'perf-baseline: %s' % error
+ write()
+ print('perf-gate: REFUSING: %s' % report['error'], flush=True)
+ return 2
report['tolerance'] = baseline['tolerance']
for note in baseline['notes']:
print('perf-gate: baseline note: %s' % note, flush=True)
@@ -758,11 +779,13 @@ def write():
failed = bool(report.get('failures'))
uncalibrated = bool(missing_baselines(report))
improved = bool(report.get('improved'))
+ stale = bool(report.get('stale_overlay'))
print('perf-gate: %s' % ('REGRESSION' if report['regression'] else
('FAILED' if failed else
('NO BASELINE' if uncalibrated else
- ('IMPROVED: REBASELINE' if improved else 'OK')))))
- return 1 if report['regression'] or failed or uncalibrated or improved else 0
+ ('IMPROVED: REBASELINE' if improved else
+ ('STALE OVERLAY' if stale else 'OK'))))))
+ return 1 if report['regression'] or failed or uncalibrated or improved or stale else 0
if __name__ == '__main__':
diff --git a/vm/selfhost/perf_baseline.py b/vm/selfhost/perf_baseline.py
index 16855bc6526..de2e4f3e95a 100644
--- a/vm/selfhost/perf_baseline.py
+++ b/vm/selfhost/perf_baseline.py
@@ -126,25 +126,34 @@ def dump(value):
return json.dumps(value, indent=1, sort_keys=True) + '\n'
-def _check_row(where, row, extra=()):
- unknown = set(row) - ROW_FIELDS - set(extra)
- if unknown:
- raise BaselineError('%s: unknown field(s) %s' % (where, ', '.join(sorted(unknown))))
+def _number(value):
+ """A real, finite number: JSON's 1e999 parses to infinity, and a bool is an int."""
+ return not isinstance(value, bool) and isinstance(value, (int, float)) and math.isfinite(value)
+
+
+def _check_ratios(where, row):
for metric in METRICS:
value = row.get(metric)
- if isinstance(value, bool) or not isinstance(value, (int, float)) or not value > 0:
+ if not _number(value) or not value > 0:
raise BaselineError('%s: %s must be a positive ratio, not %r' % (where, metric, value))
- runs = row.get('runs')
- if isinstance(runs, bool) or not isinstance(runs, int) or runs < 1:
- raise BaselineError('%s: runs must be a positive integer, not %r' % (where, runs))
tolerance = row.get('tolerance', {})
if not isinstance(tolerance, dict) or set(tolerance) - set(METRICS):
raise BaselineError('%s: tolerance must map time/memory to a fraction' % where)
for metric, value in tolerance.items():
- if isinstance(value, bool) or not isinstance(value, (int, float)) or not 0 < value < 5:
+ if not _number(value) or not 0 < value < 5:
raise BaselineError('%s: %s tolerance %r is not a fraction' % (where, metric, value))
+def _check_row(where, row, extra=()):
+ unknown = set(row) - ROW_FIELDS - set(extra)
+ if unknown:
+ raise BaselineError('%s: unknown field(s) %s' % (where, ', '.join(sorted(unknown))))
+ _check_ratios(where, row)
+ runs = row.get('runs')
+ if isinstance(runs, bool) or not isinstance(runs, int) or runs < 1:
+ raise BaselineError('%s: runs must be a positive integer, not %r' % (where, runs))
+
+
def _rows(where, tree):
"""(key, benchmark, cores, row) for every row of a {key: {bench: {cores: row}}} tree."""
if not isinstance(tree, dict):
@@ -220,10 +229,10 @@ def validate_overlay(number, overlay, where):
here = '%s rebaseline %s %s/%s' % (where, key, bench, cores)
_check_row(here, row, extra=('from',))
old = row.get('from')
- if not isinstance(old, dict) or set(old) - {'tolerance'} != set(METRICS) or \
- not isinstance(old.get('tolerance', {}), dict):
+ if not isinstance(old, dict) or set(old) - {'tolerance'} != set(METRICS):
raise BaselineError('%s: "from" must give the time and memory baseline it '
'replaces, and its tolerance if it had one' % here)
+ _check_ratios(here + ' from', old)
if rebaselines:
reason = overlay.get('reason')
if not isinstance(reason, str) or not reason.strip() or reason.strip().upper().startswith('TODO'):
@@ -299,8 +308,13 @@ def resolve(base, overlays, tolerance=None):
notes.append('%s %s/%s: already calibrated; the calibration in %s is superseded'
% (key, bench, cores, ', '.join('pr/%d.json' % n for n, _ in entries)))
continue
- rows.setdefault(key, {}).setdefault(bench, {})[cores] = _combine([r for _, r in entries],
- tolerance)
+ combined = _combine([r for _, r in entries], tolerance)
+ # Calibrations of one CPU that disagree wildly combine into a tolerance past any
+ # sane fraction: a row that gates nothing, which the fold would only refuse after
+ # the fact. Refuse it here, naming the calibrations to re-measure.
+ _check_row('%s %s/%s as combined from %s' % (
+ key, bench, cores, ', '.join('pr/%d.json' % n for n, _ in entries)), combined)
+ rows.setdefault(key, {}).setdefault(bench, {})[cores] = combined
rebaselines = {}
for number, overlay in overlays:
for key, bench, cores, row in _rows('pr/%d.json' % number, overlay.get('rebaseline', {})):
@@ -327,10 +341,12 @@ def resolve(base, overlays, tolerance=None):
return rows, notes
-def load(root=ROOT):
- """Everything perf-gate.py needs: tolerance, floor and the resolved rows."""
+def load(root=ROOT, exclude=None):
+ """Everything perf-gate.py needs: tolerance, floor and the resolved rows. `exclude` leaves
+ one pull request's overlay out (see perf-gate.py: a stale overlay of one's own)."""
policy = load_policy(root)
- rows, notes = resolve(load_base(root), load_overlays(root), policy['tolerance'])
+ rows, notes = resolve(load_base(root), [o for o in load_overlays(root) if o[0] != exclude],
+ policy['tolerance'])
return {'tolerance': policy['tolerance'], 'floor': policy['floor'], 'platforms': rows,
'notes': notes}
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index 2c49a835702..2e40970d46e 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -325,6 +325,28 @@ def test_rerunning_replaces_this_pull_requests_own_rows(self):
self.assertEqual(r['from']['time'], 1.0)
self.assertEqual(rows['linux-x64']['quicksort']['all']['time'], 0.5)
+ def test_a_stale_own_rebaseline_is_rewritten_from_fresh_runs(self):
+ # pr/7 rebaselined quicksort from 1.0; another merged change has since moved it to
+ # 0.8, so pr/7 is stale and the gate judged this run without it.
+ moved = {'linux-x64': {'quicksort': {'all': row(0.8, 0.1)}}}
+ mine = {'pr': 7, 'reason': 'first try', 'rebaseline': {'linux-x64': {'quicksort': {
+ 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}})}}}}
+ # The fresh run is inside 0.8's tolerance, so only staleness can make it rewrite.
+ overlay, rows = self.run_calibration([('linux-x64', {'quicksort': (0.82, 0.1)})],
+ existing=moved, overlays={7: mine})
+ r = overlay['rebaseline']['linux-x64']['quicksort']['all']
+ self.assertEqual(r['from'], {'time': 0.8, 'memory': 0.1})
+ self.assertEqual(rows['linux-x64']['quicksort']['all']['time'], 0.82)
+
+ def test_a_stale_row_the_runs_did_not_measure_is_reported(self):
+ moved = {'linux-x64': {'quicksort': {'all': row(0.8, 0.1)}, 'recursion': {'all': row(1.0, 0.1)}}}
+ mine = {'pr': 7, 'reason': 'first try', 'rebaseline': {'linux-x64': {'quicksort': {
+ 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}})}}}}
+ with self.assertRaises(SystemExit) as caught:
+ self.run_calibration([('linux-x64', {'recursion': (1.0, 0.1)})],
+ existing=moved, overlays={7: mine})
+ self.assertIn('quicksort', str(caught.exception))
+
def test_a_row_this_pull_request_calibrated_stays_a_calibration(self):
mine = {'pr': 7, 'calibrate': {'linux-x64@new': {'quicksort': {'all': row(1.0, 0.1, 1)}}}}
overlay, _ = self.run_calibration([('linux-x64', {'quicksort': (1.5, 0.1)}, 'New')],
@@ -377,6 +399,23 @@ def test_a_rebaseline_from_a_row_whose_tolerance_moved_is_rejected(self):
rows, _ = self.resolve([(number, overlay)], base=retuned)
self.assertEqual(rows['linux-x64@a']['quicksort']['all']['time'], 0.8)
+ def test_bad_values_are_baseline_errors_not_crashes(self):
+ number, overlay = self.rebase(12, 0.8)
+ overlay['rebaseline']['linux-x64@a']['quicksort']['all']['from']['time'] = 'fast'
+ with self.assertRaises(baselines.BaselineError):
+ baselines.validate_overlay(number, overlay, 'pr/12.json')
+ number, overlay = self.rebase(12, float('inf'))
+ with self.assertRaises(baselines.BaselineError):
+ baselines.validate_overlay(number, overlay, 'pr/12.json')
+
+ def test_wildly_disagreeing_calibrations_are_refused_not_combined(self):
+ overlays = [self.calib(12, 1.0), self.calib(13, 1.0), self.calib(14, 10.0)]
+ for n, o in overlays:
+ baselines.validate_overlay(n, o, 'pr/%d.json' % n)
+ with self.assertRaises(baselines.BaselineError) as caught:
+ baselines.resolve(self.BASE, overlays, POLICY['tolerance'])
+ self.assertIn('pr/14.json', str(caught.exception))
+
def test_a_rebaseline_needs_a_reason_and_a_from(self):
number, overlay = self.rebase(12, 0.8, reason='TODO')
with self.assertRaises(baselines.BaselineError):
@@ -542,6 +581,38 @@ def test_import_legacy_takes_only_the_branchs_own_edits(self):
tree.close()
+class StaleOverlayGateTests(unittest.TestCase):
+ """perf-gate.py with this pull request's own overlay gone stale: it must still measure
+ (here it gets as far as looking for the binaries) and record why it will fail."""
+
+ def test_the_gate_measures_without_its_own_stale_overlay(self):
+ import json
+ import os
+ moved = {'linux-x64': {'quicksort': {'all': row(0.8, 0.1)}}}
+ mine = {'pr': 7, 'reason': 'first try', 'rebaseline': {'linux-x64': {'quicksort': {
+ 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}})}}}}
+ tree = BaselineTree(moved, {7: mine})
+ saved = os.environ.get('CN1_PR_NUMBER')
+ os.environ['CN1_PR_NUMBER'] = '7'
+ try:
+ out = tree.root.parent / 'results.json'
+ gate.main(['--baseline', str(tree.root), '--out', str(out), '--platform', 'linux-x64',
+ '--binary', str(tree.root.parent / 'no-such-binary')])
+ report = json.loads(out.read_text())
+ self.assertIn('another merged change moved it first', report['stale_overlay'])
+ # It got past the baseline: what stopped it is the missing binary, not the overlay.
+ self.assertFalse(report['error'].startswith('perf-baseline'), report['error'])
+ text = gate.render_markdown(dict(report, error=None, results={}))
+ self.assertIn('perf-baseline/pr/7.json` is stale', text)
+ self.assertIn('**Result: stale overlay: re-measure required**', text)
+ finally:
+ if saved is None:
+ os.environ.pop('CN1_PR_NUMBER', None)
+ else:
+ os.environ['CN1_PR_NUMBER'] = saved
+ tree.close()
+
+
class CheckTests(unittest.TestCase):
"""perf_baseline.py check --base: what a pull request may change."""
@@ -654,6 +725,13 @@ def test_an_improvement_fails_the_job(self):
self.assertIn('IMPROVED', r.stdout)
self.assertIn('perf-baseline/pr/5931.json', r.stdout)
+ def test_a_stale_overlay_fails_the_job(self):
+ r = self.verdict({'platform': 'linux-x64', 'pr': 7, 'labels': {},
+ 'results': {'quicksort': self.row('ok')}, 'regression': False,
+ 'stale_overlay': 'pr/7.json rebaselines ... moved it first'})
+ self.assertEqual(r.returncode, 1, r.stdout + r.stderr)
+ self.assertIn('pr/7.json is stale', r.stdout)
+
def test_a_judged_run_passes(self):
r = self.verdict({'platform': 'linux-x64', 'results': {'quicksort': self.row('ok')},
'regression': False})
From 071a909e86683cd2e3f49eb3f439e4d4a7d48e55 Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 09:43:28 +0300
Subject: [PATCH 06/11] Perf baselines: a superseded own calibration is
rebaselined from the folded row
When another pull request's calibration of the same CPU was folded into base/
first, this pull request's own calibration of that row is superseded; a fresh run
past the folded row was still written as a calibration, which resolve() ignored
the same way, so the gate could never pass. Such a row is now a rebaseline from
the folded row, and writing it drops the stale calibration.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
vm/selfhost/calibrate-perf-baseline.py | 8 +++++++-
vm/selfhost/test_perf_gate.py | 13 +++++++++++++
2 files changed, 20 insertions(+), 1 deletion(-)
diff --git a/vm/selfhost/calibrate-perf-baseline.py b/vm/selfhost/calibrate-perf-baseline.py
index 6174cce4f9b..dad6611371b 100644
--- a/vm/selfhost/calibrate-perf-baseline.py
+++ b/vm/selfhost/calibrate-perf-baseline.py
@@ -164,7 +164,13 @@ def main(argv=None):
for (bench, cores), metrics in sorted(benches.items()):
current = judged.get(key, {}).get(bench, {}).get(cores)
before = others.get(key, {}).get(bench, {}).get(cores)
- new_row = current is None or cores in own.get('calibrate', {}).get(key, {}).get(bench, {})
+ # This pull request's own calibration counts only while it is the row: once
+ # another pull request's calibration of the same CPU has been folded into base/,
+ # resolve() supersedes ours, and writing a calibration again would be ignored
+ # just the same -- the gate would fail forever. Then the row is base/'s, and a
+ # move past it is a rebaseline FROM it (write_overlay drops the stale entry).
+ new_row = current is None or (
+ before is None and cores in own.get('calibrate', {}).get(key, {}).get(bench, {}))
row = {} if current is None else copy.deepcopy(current)
moved = []
for metric in METRICS:
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index 2e40970d46e..0a23ca54e81 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -347,6 +347,19 @@ def test_a_stale_row_the_runs_did_not_measure_is_reported(self):
existing=moved, overlays={7: mine})
self.assertIn('quicksort', str(caught.exception))
+ def test_a_superseded_own_calibration_becomes_a_rebaseline(self):
+ # Another pull request's calibration of the same CPU was folded into base/ first.
+ folded = {'linux-x64@new': {'quicksort': {'all': row(1.0, 0.1)}}}
+ mine = {'pr': 7, 'calibrate': {'linux-x64@new': {'quicksort': {'all': row(1.05, 0.1, 1)}}}}
+ overlay, rows = self.run_calibration([('linux-x64', {'quicksort': (2.0, 0.1)}, 'New')],
+ existing=folded, overlays={7: mine},
+ reason='quicksort got slower on purpose')
+ self.assertNotIn('calibrate', overlay)
+ r = overlay['rebaseline']['linux-x64@new']['quicksort']['all']
+ self.assertEqual((r['from']['time'], r['time']), (1.0, 2.0))
+ # ...and, unlike a second calibration, it actually moves the resolved baseline.
+ self.assertEqual(rows['linux-x64@new']['quicksort']['all']['time'], 2.0)
+
def test_a_row_this_pull_request_calibrated_stays_a_calibration(self):
mine = {'pr': 7, 'calibrate': {'linux-x64@new': {'quicksort': {'all': row(1.0, 0.1, 1)}}}}
overlay, _ = self.run_calibration([('linux-x64', {'quicksort': (1.5, 0.1)}, 'New')],
From 4ea343750c306cc4648dc0b4f08387101fa05bc7 Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 09:59:10 +0300
Subject: [PATCH 07/11] Perf baselines: rebaselines of one row chain instead of
waiting for the fold
A pull request measured on top of a merged but not yet folded rebaseline names
that one's result as its 'from', yet resolve() refused any row two overlays
touched, so it was blocked until the nightly fold. Rebaselines of a row are now
applied as a chain, each from the value the previous one left; two from the same
value (changes that never saw each other) and one whose 'from' the row never
reaches are still refused. The order comes from the 'from' values, not from
which merged first.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
vm/selfhost/perf_baseline.py | 58 ++++++++++++++++++++++-------------
vm/selfhost/test_perf_gate.py | 7 +++++
2 files changed, 43 insertions(+), 22 deletions(-)
diff --git a/vm/selfhost/perf_baseline.py b/vm/selfhost/perf_baseline.py
index de2e4f3e95a..eb0f2289a67 100644
--- a/vm/selfhost/perf_baseline.py
+++ b/vm/selfhost/perf_baseline.py
@@ -37,11 +37,13 @@
superseded and ignored: once one branch's calibration is folded, a second
branch's adds nothing and must not fail for it.
rebaseline a row that exists and moved: a deliberate change in performance, better
- or worse. "from" is the baseline the pull request measured against. If
- the row is no longer at "from" -- another merged change moved it first --
- the overlay is rejected, naming both: two changes moved the same
- benchmark, and someone has to re-measure. That is the conflict git used
- to report as a JSON hunk, now reported as what it is.
+ or worse. "from" is the baseline the pull request measured against.
+ Rebaselines of one row chain, each from the value the last one left, so a
+ pull request measured on top of a merged one need not wait for the fold.
+ Two from the SAME value, or one whose "from" the row never reaches --
+ another merged change moved it first -- are rejected, naming them: two
+ changes moved the same benchmark, and someone has to re-measure. That is
+ the conflict git used to report as a JSON hunk, now reported as what it is.
Every overlay on master belongs to a merged pull request, so `fold` needs no knowledge
of GitHub: it writes the resolved rows into base/, deletes every overlay, and refuses to
@@ -321,23 +323,35 @@ def resolve(base, overlays, tolerance=None):
rebaselines.setdefault((key, bench, cores), []).append((number, row))
for (key, bench, cores), entries in sorted(rebaselines.items()):
where = '%s %s/%s' % (key, bench, cores)
- if len(entries) > 1:
- raise BaselineError(
- '%s is rebaselined by more than one pull request (%s): two changes moved the '
- 'same benchmark. Re-measure on top of both and keep one rebaseline.'
- % (where, ', '.join('pr/%d.json' % n for n, _ in entries)))
- number, row = entries[0]
- current = rows.get(key, {}).get(bench, {}).get(cores)
- if current is None:
- raise BaselineError('pr/%d.json rebaselines %s, which has no row; a new row is a '
- '"calibrate" entry' % (number, where))
- if not _same(row['from'], current):
- raise BaselineError(
- 'pr/%d.json rebaselines %s from %s, but the row is now %s: another merged '
- 'change moved it first. Re-measure on top of it and update the rebaseline.'
- % (number, where, json.dumps(row['from'], sort_keys=True),
- json.dumps(from_row(current), sort_keys=True)))
- rows[key][bench][cores] = {k: v for k, v in row.items() if k != 'from'}
+ # Several rebaselines of one row are applied as a CHAIN, each from the value the
+ # previous one left: a pull request that measured on top of one merged but not yet
+ # folded names that one's result as its "from", and must not wait for the nightly
+ # fold. Two entries from the SAME value are two changes that each moved the row
+ # without seeing the other -- the real conflict. The order comes from the "from"
+ # values alone, so it does not depend on which merged first.
+ pending = list(entries)
+ while pending:
+ current = rows.get(key, {}).get(bench, {}).get(cores)
+ if current is None:
+ raise BaselineError('pr/%d.json rebaselines %s, which has no row; a new row is '
+ 'a "calibrate" entry' % (pending[0][0], where))
+ ready = [(n, r) for n, r in pending if _same(r['from'], current)]
+ if len(ready) > 1:
+ raise BaselineError(
+ '%s is rebaselined from the same value by more than one pull request (%s): '
+ 'two changes moved the same benchmark without seeing each other. '
+ 'Re-measure on top of both and keep one rebaseline.'
+ % (where, ', '.join('pr/%d.json' % n for n, _ in ready)))
+ if not ready:
+ number, row = pending[0]
+ raise BaselineError(
+ 'pr/%d.json rebaselines %s from %s, but the row is now %s: another merged '
+ 'change moved it first. Re-measure on top of it and update the rebaseline.'
+ % (number, where, json.dumps(row['from'], sort_keys=True),
+ json.dumps(from_row(current), sort_keys=True)))
+ number, row = ready[0]
+ rows[key][bench][cores] = {k: v for k, v in row.items() if k != 'from'}
+ pending.remove(ready[0])
return rows, notes
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index 0a23ca54e81..50c4f48f463 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -389,6 +389,13 @@ def test_a_rebaseline_replaces_the_row(self):
rows, _ = self.resolve([self.rebase(12, 0.8)])
self.assertEqual(rows['linux-x64@a']['quicksort']['all'], row(0.8, 0.1))
+ def test_a_rebaseline_measured_on_top_of_an_unfolded_one_chains(self):
+ # pr/12 merged (1.0 -> 0.8) and is not folded yet; pr/15 measured on top of it.
+ for order in ([self.rebase(12, 0.8), self.rebase(15, 0.6, frm=0.8)],
+ [self.rebase(15, 0.6, frm=0.8), self.rebase(12, 0.8)]):
+ rows, _ = self.resolve(order)
+ self.assertEqual(rows['linux-x64@a']['quicksort']['all']['time'], 0.6)
+
def test_two_rebaselines_of_one_row_are_a_conflict_naming_both(self):
with self.assertRaises(baselines.BaselineError) as caught:
self.resolve([self.rebase(12, 0.8), self.rebase(15, 0.9)])
From 1e4a8971da435c76ddcf341e11f29dfec0f0be1a Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 11:06:29 +0300
Subject: [PATCH 08/11] Perf baselines: refuse calibrating an undecodable CPU
beside per-model rows
baseline_key() selects no row for a runner whose CPU model cannot be decoded on
a platform with per-model rows, so the plain-platform row the calibrator used to
write there could never be read and the gate stayed uncalibrated. It now refuses
and points at perf-gate.cpu_model(). The fold's one-shot site rebuild is kept,
with a comment on why: website-docs.yml rebuilds daily on its own.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
.github/workflows/perf-baseline.yml | 6 ++++++
vm/selfhost/calibrate-perf-baseline.py | 10 ++++++++++
vm/selfhost/test_perf_gate.py | 7 +++++++
3 files changed, 23 insertions(+)
diff --git a/.github/workflows/perf-baseline.yml b/.github/workflows/perf-baseline.yml
index 5ad54b50169..f8a7a92c440 100644
--- a/.github/workflows/perf-baseline.yml
+++ b/.github/workflows/perf-baseline.yml
@@ -104,6 +104,12 @@ jobs:
echo "Could not push the fold after 3 attempts." >&2
exit 1
+ # One-shot on purpose. If this dispatch fails after the fold was pushed, a re-run
+ # finds nothing to fold and skips it -- and that is acceptable: website-docs.yml
+ # also rebuilds on its own daily schedule (and port-status-nightly dispatches it),
+ # so the JDK 25 table, which shows gate baselines rather than live results, is at
+ # most a day behind. Persisting a "rebuild owed" flag to close that window would
+ # add state for no visible benefit.
- name: Rebuild the Port Status page
if: steps.fold.outputs.folded == 'true'
env:
diff --git a/vm/selfhost/calibrate-perf-baseline.py b/vm/selfhost/calibrate-perf-baseline.py
index dad6611371b..8bca4b08d1d 100644
--- a/vm/selfhost/calibrate-perf-baseline.py
+++ b/vm/selfhost/calibrate-perf-baseline.py
@@ -97,6 +97,16 @@ def collect(paths, only=None, rows=None):
cls = perf_gate.cpu_class(report.get('cpu'))
key = perf_gate.baseline_key(rows or {}, report['platform'], report.get('cpu'))
if key is None or key not in (rows or {}):
+ if cls is None and any(k.startswith(report['platform'] + '@') for k in rows or {}):
+ # An undecodable CPU on a platform with per-model rows: baseline_key() selects
+ # NO row for it (judging an unknown model by another model's ratios is what
+ # failed unchanged code), so a plain-platform row written here would never
+ # be read and the gate would stay uncalibrated forever. The fix is to teach
+ # perf-gate.cpu_model() to read this runner, not to calibrate.
+ raise SystemExit('%s: the runner reported no decodable CPU model (%r) on %s, '
+ 'which has per-model rows; no baseline row can be selected '
+ 'for it. Fix perf-gate.cpu_model() for this runner.'
+ % (path, report.get('cpu'), report['platform']))
key = '%s@%s' % (report['platform'], cls) if cls else report['platform']
for bench, by_cores in report['results'].items():
if only and bench not in only:
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index 50c4f48f463..6536d57bc7f 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -271,6 +271,13 @@ def test_a_run_feeds_the_row_that_judged_it(self):
self.assertNotIn('calibrate', overlay)
self.assertEqual(sorted(rows), ['macos-arm64'])
+ def test_an_undecodable_cpu_cannot_be_calibrated_beside_per_model_rows(self):
+ old = {'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1)}}}
+ with self.assertRaises(SystemExit) as caught:
+ self.run_calibration([('linux-x64', {'quicksort': (1.2, 0.1)}, 'unknown')],
+ existing=old)
+ self.assertIn('cpu_model', str(caught.exception))
+
def test_a_row_inside_its_tolerance_is_left_alone(self):
old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1)},
'recursion': {'all': row(1.0, 0.1)}}}
From 79c7a685e88a9e974526eb93b2b6806ac22fb617 Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 11:31:14 +0300
Subject: [PATCH 09/11] Website build runs when the JDK 25 table's renderer
changes
build.sh renders the Port Status JDK 25 table with perf_baseline.py summary, but
website-docs.yml's path filters did not include it, so a change to its output
shape skipped the Hugo build and validate_port_status until a scheduled
production build. The renderer is now a trigger; the baseline data is not (a row
changes numbers, not shape, and the nightly fold dispatches a rebuild).
Comments record why two recovery paths are deliberately absent: recalibrating a
rebaseline another overlay is chained on would make that one stale, and a CPU
calibrated an order of magnitude apart is a broken measurement to delete.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
.github/workflows/website-docs.yml | 7 +++++++
vm/selfhost/calibrate-perf-baseline.py | 4 ++++
vm/selfhost/perf_baseline.py | 6 +++++-
3 files changed, 16 insertions(+), 1 deletion(-)
diff --git a/.github/workflows/website-docs.yml b/.github/workflows/website-docs.yml
index 8b1353f71cb..079e677b2c5 100644
--- a/.github/workflows/website-docs.yml
+++ b/.github/workflows/website-docs.yml
@@ -45,6 +45,11 @@ on:
- 'vm/backend/src/**'
- 'vm/backend/impl/**'
- 'vm/backend/shared-sources.sh'
+ # build.sh renders the Port Status JDK 25 table with this script's `summary`, so a
+ # change to its output shape must go through the Hugo build and validate_port_status.
+ # The baseline DATA (vm/selfhost/perf-baseline/**) is left out on purpose: a row
+ # changes numbers, not shape, and the nightly fold dispatches a rebuild for it.
+ - 'vm/selfhost/perf_baseline.py'
- '.github/workflows/website-docs.yml'
push:
branches: [main, master]
@@ -78,6 +83,8 @@ on:
- 'vm/backend/src/**'
- 'vm/backend/impl/**'
- 'vm/backend/shared-sources.sh'
+ # As above: the JDK 25 table's renderer.
+ - 'vm/selfhost/perf_baseline.py'
- '.github/workflows/website-docs.yml'
workflow_dispatch:
inputs:
diff --git a/vm/selfhost/calibrate-perf-baseline.py b/vm/selfhost/calibrate-perf-baseline.py
index 8bca4b08d1d..2ed234e2764 100644
--- a/vm/selfhost/calibrate-perf-baseline.py
+++ b/vm/selfhost/calibrate-perf-baseline.py
@@ -142,6 +142,10 @@ def main(argv=None):
# What the gate judged against (this pull request's earlier rows included), and the
# baseline as it stands without them -- which is what a rebaseline's "from" names,
# since this run replaces this pull request's earlier rebaseline rather than stacking.
+ # This raises when another overlay chains its rebaseline ON TOP of this pull request's
+ # (its "from" is our value), and that is deliberate rather than a gap: re-measuring a
+ # link something else was measured on would move the value that later link starts
+ # from and make IT stale. Recalibrate a pull request nothing has built on yet.
others, _ = perf_baseline.resolve(base, [o for o in overlays if o[0] != number], tolerance)
try:
judged, _ = perf_baseline.resolve(base, overlays, tolerance)
diff --git a/vm/selfhost/perf_baseline.py b/vm/selfhost/perf_baseline.py
index eb0f2289a67..24fa4089acf 100644
--- a/vm/selfhost/perf_baseline.py
+++ b/vm/selfhost/perf_baseline.py
@@ -313,7 +313,11 @@ def resolve(base, overlays, tolerance=None):
combined = _combine([r for _, r in entries], tolerance)
# Calibrations of one CPU that disagree wildly combine into a tolerance past any
# sane fraction: a row that gates nothing, which the fold would only refuse after
- # the fact. Refuse it here, naming the calibrations to re-measure.
+ # the fact. Refuse it here, naming the calibrations. There is deliberately no
+ # automatic recovery (the calibrator does not rewrite a pull request's calibrate
+ # entry when this fires): one CPU measured an order of magnitude apart by runs of
+ # the same code is a broken measurement, and the fix is to delete that entry from
+ # the pr/.json this message names, not to re-measure around it.
_check_row('%s %s/%s as combined from %s' % (
key, bench, cores, ', '.join('pr/%d.json' % n for n, _ in entries)), combined)
rows.setdefault(key, {}).setdefault(bench, {})[cores] = combined
From ba36b34d61b13964c32de573e622d4e6a7d6ba50 Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 11:56:42 +0300
Subject: [PATCH 10/11] Perf baselines: a reason per rebaselined row; per-cell
CPU coverage in the table
The overlay carried ONE reason for the whole file, so once a pull request had
written a rebaseline, any row it rebaselined later passed under that earlier
row's explanation -- a regression accepted for one cause, explained by another.
The reason now belongs to each rebaseline row, like its 'from': required and
validated per row, reused only when that same row is re-measured, never
inherited by a row new to the overlay; a file-level reason is refused.
pr/5930.json gives each of its four rows its own.
The JDK 25 summary recorded one CPU list per platform while each cell's median
could cover fewer models (one calibrated for some benchmarks only), so published
numbers could shift as models were filled in with nothing to show it. Each cell
now records the models it covers, and its tooltip says how many of the
platform's.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
.../website/layouts/_default/port-status.html | 3 +-
vm/selfhost/calibrate-perf-baseline.py | 17 +++--
vm/selfhost/perf-baseline/pr/5930.json | 5 +-
vm/selfhost/perf_baseline.py | 54 ++++++++++-----
vm/selfhost/test_perf_gate.py | 67 +++++++++++++++----
5 files changed, 111 insertions(+), 35 deletions(-)
diff --git a/docs/website/layouts/_default/port-status.html b/docs/website/layouts/_default/port-status.html
index 132241a93ab..c01640aeb8d 100644
--- a/docs/website/layouts/_default/port-status.html
+++ b/docs/website/layouts/_default/port-status.html
@@ -351,7 +351,8 @@ ParparVM against JDK 25
{{- range $jdk.platforms }}
{{- $platform := . -}}
{{- $cell := index .benchmarks $benchmark.id -}}
-
+ {{- $covered := 0 -}}{{- with $cell -}}{{- $covered = len .cpus -}}{{- end -}}
+
{{- with $cell -}}
{{- range $metric := slice "time" "memory" -}}
{{- $m := index $cell $metric -}}
diff --git a/vm/selfhost/calibrate-perf-baseline.py b/vm/selfhost/calibrate-perf-baseline.py
index 2ed234e2764..084f2ba18fb 100644
--- a/vm/selfhost/calibrate-perf-baseline.py
+++ b/vm/selfhost/calibrate-perf-baseline.py
@@ -17,7 +17,8 @@
rebaseline for a row the runs put outside its tolerance, either way: a change that
moved performance on purpose. Only the metric that moved is replaced, and
--reason is required -- the reason is what a reviewer reads beside the
- number. --all rebaselines every measured row instead, for recalibrating a
+ number; a row this pull request already rebaselined keeps its reason when
+ re-measured. --all rebaselines every measured row instead, for recalibrating a
row's noise from several runs of unchanged code; --metric limits that to one
metric, so recalibrating a noisy RAM figure leaves a steady time alone.
@@ -226,9 +227,17 @@ def main(argv=None):
if not calibrate and not rebaseline:
print('Every measured row is inside its tolerance; nothing to write.')
return 0
- if rebaseline and not (args.reason or own.get('reason')):
- raise SystemExit('These rows moved past their tolerance and would be rebaselined; say '
- 'why with --reason:\n' + json.dumps(rebaseline, indent=1))
+ if not args.reason:
+ # A row this overlay already rebaselined keeps its reason when re-measured; a row
+ # NEW to it needs its own -- never an earlier row's explanation by default.
+ unexplained = ['%s %s/%s' % (key, bench, cores)
+ for key, bench, cores, _ in perf_baseline._rows('new', rebaseline)
+ if not own.get('rebaseline', {}).get(key, {}).get(bench, {})
+ .get(cores, {}).get('reason')]
+ if unexplained:
+ raise SystemExit('These rows moved past their tolerance and would be rebaselined; '
+ 'say why with --reason: %s\n%s' % (', '.join(unexplained),
+ json.dumps(rebaseline, indent=1)))
try:
path = perf_baseline.write_overlay(root, number, calibrate, rebaseline, args.reason)
perf_baseline.load(root) # the overlay must resolve against everything else
diff --git a/vm/selfhost/perf-baseline/pr/5930.json b/vm/selfhost/perf-baseline/pr/5930.json
index 83e61d1964b..3910672c488 100644
--- a/vm/selfhost/perf-baseline/pr/5930.json
+++ b/vm/selfhost/perf-baseline/pr/5930.json
@@ -1,6 +1,5 @@
{
"pr": 5930,
- "reason": "Rows whose readings are bimodal or wide on these runners with unchanged VM code, calibrated from runs that all landed on one side, so the other side failed once improvements fail the gate. Recalibrated from all of each runner's CI runs, one metric each, the other left as it was: stringBuilding RAM on Windows x64 AMD Family 25 Model 1 (0.267x or 0.322x, 10 runs) and on macOS arm64 (0.389x-0.54x, 5 runs); translator time on that Windows CPU (rounds alternate 0.42x-0.49x and 0.59x-0.64x inside single runs; run medians 0.467x-0.628x, 11 runs) and hello time there (rounds 0.87x-1.57x, run medians 1.051x-1.352x, 11 runs).",
"rebaseline": {
"macos-arm64": {
"stringBuilding": {
@@ -13,6 +12,7 @@
}
},
"memory": 0.463,
+ "reason": "RAM reads 0.389x-0.54x on this runner with unchanged VM code across 5 CI runs, and the row was calibrated at the top of that range. RAM recalibrated from all 5; time untouched.",
"runs": 5,
"time": 0.842,
"tolerance": {
@@ -30,6 +30,7 @@
"time": 1.241
},
"memory": 0.803,
+ "reason": "Time is wide on this runner with unchanged VM code: rounds 0.87x-1.57x, run medians 1.051x-1.352x across 11 runs, and a master run had already fallen outside the 15% band. Time recalibrated from all 11; RAM untouched.",
"runs": 11,
"time": 1.228,
"tolerance": {
@@ -44,6 +45,7 @@
"time": 1.216
},
"memory": 0.322,
+ "reason": "RAM is bimodal on this runner with unchanged VM code: 0.267x or 0.322x across 10 CI runs, and the row was calibrated from runs on the upper mode. RAM recalibrated from all 10; time untouched.",
"runs": 10,
"time": 1.216,
"tolerance": {
@@ -58,6 +60,7 @@
"time": 0.606
},
"memory": 0.542,
+ "reason": "Time is bimodal on this runner with unchanged VM code: rounds alternate 0.42x-0.49x and 0.59x-0.64x inside single runs, so run medians land on either mode (0.467x-0.628x across 11 runs). Time recalibrated from all 11; RAM untouched.",
"runs": 11,
"time": 0.605,
"tolerance": {
diff --git a/vm/selfhost/perf_baseline.py b/vm/selfhost/perf_baseline.py
index 24fa4089acf..5a20aeeaeb0 100644
--- a/vm/selfhost/perf_baseline.py
+++ b/vm/selfhost/perf_baseline.py
@@ -25,9 +25,9 @@
AN OVERLAY (pr/.json)
{"pr": 5931,
- "reason": "why any rebaselined row moved (required when there are any)",
"calibrate": {key: {benchmark: {cores: row}}},
- "rebaseline": {key: {benchmark: {cores: row + "from": {"time": t, "memory": m}}}}}
+ "rebaseline": {key: {benchmark: {cores: row + "from": {"time": t, "memory": m, ...}
+ + "reason": "why THIS row moved"}}}}
calibrate a row that does not exist yet: a runner CPU model no run had measured.
Hardware, not code -- whichever branch met the runner first carries it.
@@ -37,7 +37,11 @@
superseded and ignored: once one branch's calibration is folded, a second
branch's adds nothing and must not fail for it.
rebaseline a row that exists and moved: a deliberate change in performance, better
- or worse. "from" is the baseline the pull request measured against.
+ or worse. "from" is the baseline the pull request measured against, and
+ "reason" says why this row moved -- per row, because a pull request can
+ rebaseline rows at different times for different causes, and one reason
+ for the whole file let a later row pass under an earlier row's
+ explanation.
Rebaselines of one row chain, each from the value the last one left, so a
pull request measured on top of a merged one need not wait for the fold.
Two from the SAME value, or one whose "from" the row never reaches --
@@ -219,7 +223,11 @@ def load_overlays(root=ROOT):
def validate_overlay(number, overlay, where):
if not isinstance(overlay, dict):
raise BaselineError('%s must be an object' % where)
- unknown = set(overlay) - {'pr', 'reason', 'calibrate', 'rebaseline'}
+ if 'reason' in overlay:
+ raise BaselineError('%s: "reason" belongs to each rebaseline row, not the file: one '
+ 'reason for the file lets a later row pass under an earlier '
+ 'row\'s explanation' % where)
+ unknown = set(overlay) - {'pr', 'calibrate', 'rebaseline'}
if unknown:
raise BaselineError('%s: unknown field(s) %s' % (where, ', '.join(sorted(unknown))))
if overlay.get('pr') != number:
@@ -229,17 +237,17 @@ def validate_overlay(number, overlay, where):
rebaselines = list(_rows(where + ' rebaseline', overlay.get('rebaseline', {})))
for key, bench, cores, row in rebaselines:
here = '%s rebaseline %s %s/%s' % (where, key, bench, cores)
- _check_row(here, row, extra=('from',))
+ _check_row(here, row, extra=('from', 'reason'))
+ reason = row.get('reason')
+ if not isinstance(reason, str) or not reason.strip() or \
+ reason.strip().upper().startswith('TODO'):
+ raise BaselineError('%s: a rebaseline needs a "reason" saying why this row moved'
+ % here)
old = row.get('from')
if not isinstance(old, dict) or set(old) - {'tolerance'} != set(METRICS):
raise BaselineError('%s: "from" must give the time and memory baseline it '
'replaces, and its tolerance if it had one' % here)
_check_ratios(here + ' from', old)
- if rebaselines:
- reason = overlay.get('reason')
- if not isinstance(reason, str) or not reason.strip() or reason.strip().upper().startswith('TODO'):
- raise BaselineError('%s: a rebaseline needs a "reason" saying why the rows moved'
- % where)
if not overlay.get('calibrate') and not rebaselines:
raise BaselineError('%s changes nothing; delete it' % where)
@@ -354,7 +362,8 @@ def resolve(base, overlays, tolerance=None):
% (number, where, json.dumps(row['from'], sort_keys=True),
json.dumps(from_row(current), sort_keys=True)))
number, row = ready[0]
- rows[key][bench][cores] = {k: v for k, v in row.items() if k != 'from'}
+ rows[key][bench][cores] = {k: v for k, v in row.items()
+ if k not in ('from', 'reason')}
pending.remove(ready[0])
return rows, notes
@@ -400,11 +409,20 @@ def fold(root=ROOT):
def write_overlay(root, number, calibrate=None, rebaseline=None, reason=None):
"""Add rows to pr/.json, creating it if needed. A row written again replaces
- the earlier one, so re-running the calibration after another CI round is safe."""
+ the earlier one, so re-running the calibration after another CI round is safe.
+
+ `reason` is stamped on every rebaseline row written that does not carry its own. A
+ row re-measured without one keeps the reason it already had -- the same row, moved for
+ the same cause -- but a row new to this overlay gets no inherited explanation: with
+ neither, validation fails and says so."""
path = Path(root) / 'pr' / ('%d.json' % number)
overlay = _read_json(path) if path.exists() else {'pr': number}
for kind, tree in (('calibrate', calibrate), ('rebaseline', rebaseline)):
for key, bench, cores, row in _rows(kind, tree or {}):
+ if kind == 'rebaseline' and 'reason' not in row:
+ earlier = overlay.get('rebaseline', {}).get(key, {}).get(bench, {}).get(cores, {})
+ if reason or earlier.get('reason'):
+ row = dict(row, reason=reason or earlier['reason'])
overlay.setdefault(kind, {}).setdefault(key, {}).setdefault(bench, {})[cores] = row
# One row is one kind: a row calibrated earlier on this branch and now moved
# again is still a calibration, and a stale entry of the other kind would
@@ -421,8 +439,6 @@ def write_overlay(root, number, calibrate=None, rebaseline=None, reason=None):
del tree[key]
if kind in overlay and not tree:
del overlay[kind]
- if reason:
- overlay['reason'] = reason
validate_overlay(number, overlay, path.name)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(dump(overlay))
@@ -480,8 +496,12 @@ def summary(rows):
for k in keys),
'benchmarks': {}}
for bench, _, _ in BENCHMARKS:
- values = [benches[bench]['all'] for benches in keys.values()
- if 'all' in benches.get(bench, {})]
+ # Which CPU models THIS cell covers. A model can be calibrated for some
+ # benchmarks and not yet others, and a cell's median taken over a different set
+ # than the platform's list would shift silently as models are filled in.
+ contributing = sorted((k, benches[bench]['all']) for k, benches in keys.items()
+ if 'all' in benches.get(bench, {}))
+ values = [v for _, v in contributing]
if not values:
continue
entry['benchmarks'][bench] = {
@@ -489,6 +509,8 @@ def summary(rows):
'min': min(v[metric] for v in values),
'max': max(v[metric] for v in values)}
for metric in METRICS}
+ entry['benchmarks'][bench]['cpus'] = [
+ k.split('@', 1)[1] if '@' in k else 'unidentified CPU' for k, _ in contributing]
out.append(entry)
return {'schema_version': 1,
'source': 'vm/selfhost/perf-baseline',
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index 6536d57bc7f..3a3b162914f 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -299,7 +299,7 @@ def test_a_moved_row_is_rebaselined_from_its_old_value_and_needs_a_reason(self):
# Only the metric that moved is replaced; RAM keeps its baseline and tolerance.
self.assertEqual(r['memory'], 0.1)
self.assertEqual(r['tolerance'], {'memory': 0.3})
- self.assertEqual(overlay['reason'], 'faster partitioning')
+ self.assertEqual(r['reason'], 'faster partitioning')
self.assertEqual(rows['linux-x64']['quicksort']['all']['time'], 0.7)
def test_all_recalibrates_every_measured_row(self):
@@ -321,10 +321,34 @@ def test_metric_limits_a_recalibration_to_one_metric(self):
self.assertEqual(r['memory'], 0.32)
self.assertGreaterEqual(r['tolerance']['memory'], 0.25) # covers the 0.27 mode
+ def test_a_new_row_never_inherits_another_rows_reason(self):
+ old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1)}, 'recursion': {'all': row(1.0, 0.1)}}}
+ mine = {'pr': 7, 'rebaseline': {'linux-x64': {'quicksort': {'all': dict(
+ row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}, 'reason': 'faster sort'})}}}}
+ # recursion moved too, later, for some other cause: "faster sort" must not cover it.
+ with self.assertRaises(SystemExit) as caught:
+ self.run_calibration([('linux-x64', {'recursion': (1.5, 0.1)})],
+ existing=old, overlays={7: mine})
+ self.assertIn('recursion', str(caught.exception))
+ overlay, _ = self.run_calibration([('linux-x64', {'recursion': (1.5, 0.1)})],
+ existing=old, overlays={7: mine}, reason='slower calls')
+ rows = overlay['rebaseline']['linux-x64']
+ self.assertEqual((rows['quicksort']['all']['reason'], rows['recursion']['all']['reason']),
+ ('faster sort', 'slower calls'))
+
+ def test_a_re_measured_row_keeps_its_own_reason(self):
+ old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1)}}}
+ mine = {'pr': 7, 'rebaseline': {'linux-x64': {'quicksort': {'all': dict(
+ row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}, 'reason': 'faster sort'})}}}}
+ overlay, _ = self.run_calibration([('linux-x64', {'quicksort': (0.5, 0.1)})],
+ existing=old, overlays={7: mine})
+ self.assertEqual(overlay['rebaseline']['linux-x64']['quicksort']['all']['reason'],
+ 'faster sort')
+
def test_rerunning_replaces_this_pull_requests_own_rows(self):
old = {'linux-x64': {'quicksort': {'all': row(1.0, 0.1)}}}
- mine = {'pr': 7, 'reason': 'first try', 'rebaseline': {'linux-x64': {'quicksort': {
- 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}})}}}}
+ mine = {'pr': 7, 'rebaseline': {'linux-x64': {'quicksort': {
+ 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}, 'reason': 'first try'})}}}}
overlay, rows = self.run_calibration([('linux-x64', {'quicksort': (0.5, 0.1)})],
existing=old, overlays={7: mine})
r = overlay['rebaseline']['linux-x64']['quicksort']['all']
@@ -336,8 +360,8 @@ def test_a_stale_own_rebaseline_is_rewritten_from_fresh_runs(self):
# pr/7 rebaselined quicksort from 1.0; another merged change has since moved it to
# 0.8, so pr/7 is stale and the gate judged this run without it.
moved = {'linux-x64': {'quicksort': {'all': row(0.8, 0.1)}}}
- mine = {'pr': 7, 'reason': 'first try', 'rebaseline': {'linux-x64': {'quicksort': {
- 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}})}}}}
+ mine = {'pr': 7, 'rebaseline': {'linux-x64': {'quicksort': {
+ 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}, 'reason': 'first try'})}}}}
# The fresh run is inside 0.8's tolerance, so only staleness can make it rewrite.
overlay, rows = self.run_calibration([('linux-x64', {'quicksort': (0.82, 0.1)})],
existing=moved, overlays={7: mine})
@@ -347,8 +371,8 @@ def test_a_stale_own_rebaseline_is_rewritten_from_fresh_runs(self):
def test_a_stale_row_the_runs_did_not_measure_is_reported(self):
moved = {'linux-x64': {'quicksort': {'all': row(0.8, 0.1)}, 'recursion': {'all': row(1.0, 0.1)}}}
- mine = {'pr': 7, 'reason': 'first try', 'rebaseline': {'linux-x64': {'quicksort': {
- 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}})}}}}
+ mine = {'pr': 7, 'rebaseline': {'linux-x64': {'quicksort': {
+ 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}, 'reason': 'first try'})}}}}
with self.assertRaises(SystemExit) as caught:
self.run_calibration([('linux-x64', {'recursion': (1.0, 0.1)})],
existing=moved, overlays={7: mine})
@@ -386,8 +410,8 @@ def resolve(self, overlays, base=None):
return baselines.resolve(base if base is not None else self.BASE, overlays)
def rebase(self, number, t, frm=1.0, reason='moved'):
- return (number, {'pr': number, 'reason': reason, 'rebaseline': {'linux-x64@a': {
- 'quicksort': {'all': dict(row(t, 0.1), **{'from': {'time': frm, 'memory': 0.1}})}}}})
+ return (number, {'pr': number, 'rebaseline': {'linux-x64@a': {'quicksort': {
+ 'all': dict(row(t, 0.1), **{'from': {'time': frm, 'memory': 0.1}, 'reason': reason})}}}})
def calib(self, number, t, key='linux-x64@b', runs=1, **tol):
return (number, {'pr': number, 'calibrate': {key: {'quicksort': {'all': row(t, 0.2, runs, **tol)}}}})
@@ -443,6 +467,13 @@ def test_wildly_disagreeing_calibrations_are_refused_not_combined(self):
baselines.resolve(self.BASE, overlays, POLICY['tolerance'])
self.assertIn('pr/14.json', str(caught.exception))
+ def test_a_file_level_reason_is_refused(self):
+ number, overlay = self.rebase(12, 0.8)
+ overlay['reason'] = 'covers everything'
+ with self.assertRaises(baselines.BaselineError) as caught:
+ baselines.validate_overlay(number, overlay, 'pr/12.json')
+ self.assertIn('each rebaseline row', str(caught.exception))
+
def test_a_rebaseline_needs_a_reason_and_a_from(self):
number, overlay = self.rebase(12, 0.8, reason='TODO')
with self.assertRaises(baselines.BaselineError):
@@ -488,8 +519,8 @@ def test_a_calibration_of_an_existing_row_is_superseded_not_fatal(self):
self.assertIn('superseded', notes[0])
def test_a_rebaseline_can_move_a_row_another_branch_calibrated(self):
- frm = (15, {'pr': 15, 'reason': 'moved', 'rebaseline': {'linux-x64@b': {'quicksort': {
- 'all': dict(row(0.5, 0.2), **{'from': {'time': 1.0, 'memory': 0.2}})}}}})
+ frm = (15, {'pr': 15, 'rebaseline': {'linux-x64@b': {'quicksort': {
+ 'all': dict(row(0.5, 0.2), **{'from': {'time': 1.0, 'memory': 0.2}, 'reason': 'moved'})}}}})
rows, _ = self.resolve([self.calib(12, 1.0), frm])
self.assertEqual(rows['linux-x64@b']['quicksort']['all']['time'], 0.5)
@@ -616,8 +647,8 @@ def test_the_gate_measures_without_its_own_stale_overlay(self):
import json
import os
moved = {'linux-x64': {'quicksort': {'all': row(0.8, 0.1)}}}
- mine = {'pr': 7, 'reason': 'first try', 'rebaseline': {'linux-x64': {'quicksort': {
- 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}})}}}}
+ mine = {'pr': 7, 'rebaseline': {'linux-x64': {'quicksort': {
+ 'all': dict(row(0.7, 0.1), **{'from': {'time': 1.0, 'memory': 0.1}, 'reason': 'first try'})}}}}
tree = BaselineTree(moved, {7: mine})
saved = os.environ.get('CN1_PR_NUMBER')
os.environ['CN1_PR_NUMBER'] = '7'
@@ -693,6 +724,16 @@ def test_one_entry_per_platform_with_the_spread_of_its_cpus(self):
self.assertEqual(linux['benchmarks']['quicksort']['time'],
{'median': 1.2, 'min': 1.0, 'max': 1.4})
+ def test_each_cell_names_the_cpus_it_covers(self):
+ # Model b is calibrated for quicksort only so far.
+ rows = {'linux-x64@a': {'quicksort': {'all': row(1.0, 0.1)}, 'recursion': {'all': row(2.0, 0.1)}},
+ 'linux-x64@b': {'quicksort': {'all': row(1.4, 0.3)}}}
+ linux = baselines.summary(rows)['platforms'][0]
+ self.assertEqual(linux['cpus'], ['a', 'b'])
+ self.assertEqual(linux['benchmarks']['quicksort']['cpus'], ['a', 'b'])
+ self.assertEqual(linux['benchmarks']['recursion']['cpus'], ['a'])
+ self.assertEqual(linux['benchmarks']['recursion']['time']['median'], 2.0)
+
def test_every_gated_benchmark_is_described(self):
self.assertEqual([b for b, _, _ in baselines.BENCHMARKS],
['hello', 'translator'] + gate.WORKLOADS)
From 9526f92808a961c963cab81aad55c79df7ad4a6e Mon Sep 17 00:00:00 2001
From: Shai Almog <67850168+shai-almog@users.noreply.github.com>
Date: Fri, 2 Oct 2026 13:45:02 +0300
Subject: [PATCH 11/11] Perf gate: the calibration command carries --reason
when the run also moved a row
The calibrator processes the whole results file, so a run that both fills a
missing row and moves another past its tolerance writes a calibration AND a
rebaseline -- and refuses without a reason for the rebaseline. The verdict step
and the PR comment printed the reason-less command in that case; both now print
the one that works.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
vm/selfhost/ci-perf-gate.sh | 7 ++++++-
vm/selfhost/perf-gate.py | 4 +++-
vm/selfhost/test_perf_gate.py | 17 +++++++++++++++++
3 files changed, 26 insertions(+), 2 deletions(-)
diff --git a/vm/selfhost/ci-perf-gate.sh b/vm/selfhost/ci-perf-gate.sh
index d4cc55ebdab..aa3f95f09c0 100755
--- a/vm/selfhost/ci-perf-gate.sh
+++ b/vm/selfhost/ci-perf-gate.sh
@@ -73,7 +73,12 @@ if missing or report.get('calibration'):
print('ParparVM performance gate: NO BASELINE on %s for %s (runner CPU: %s).'
% (report['platform'], key, report.get('cpu', 'unknown')))
print('Add it to %s from this job\'s perf-results.json:' % overlay)
- print(' python3 vm/selfhost/calibrate-perf-baseline.py --pr %s perf-results.json' % number)
+ # The calibrator processes the whole file, so if this run ALSO moved a row past its
+ # tolerance it writes that rebaseline too -- and refuses without a reason for it.
+ if moved('improved'):
+ print(rebaseline)
+ else:
+ print(' python3 vm/selfhost/calibrate-perf-baseline.py --pr %s perf-results.json' % number)
print(json.dumps({key: report.get('calibration') or sorted(missing)}, indent=1))
sys.exit(1)
improved = moved('improved')
diff --git a/vm/selfhost/perf-gate.py b/vm/selfhost/perf-gate.py
index 5cfd137b6c1..87a69ba2251 100644
--- a/vm/selfhost/perf-gate.py
+++ b/vm/selfhost/perf-gate.py
@@ -565,7 +565,9 @@ def render_markdown(report):
'commit it with this pull request:'
% (count, '' if count == 1 else 's', target, 'it' if count == 1 else 'them',
overlay_name(report)), '',
- '```', fix_command(report, False), '```',
+ # The calibrator takes the whole file: if this run also moved a row,
+ # that rebaseline needs a reason or the command is refused.
+ '```', fix_command(report, bool(regressions or improvements)), '```',
'', 'The rows it will add ', '',
'```json', json.dumps({target: report['calibration']}, indent=1),
'```', '', ' ']
diff --git a/vm/selfhost/test_perf_gate.py b/vm/selfhost/test_perf_gate.py
index 3a3b162914f..5494e96c3cf 100644
--- a/vm/selfhost/test_perf_gate.py
+++ b/vm/selfhost/test_perf_gate.py
@@ -109,6 +109,15 @@ def test_a_calibration_names_this_pull_requests_overlay(self):
self.assertIn('pr/.json', text)
self.assertIn('calibrate-perf-baseline.py --pr perf-results.json', text)
+ def test_a_calibration_beside_an_improvement_asks_for_a_reason(self):
+ r = report({'quicksort': {'all': (metric(1.3, None, 'uncalibrated'),
+ metric(0.9, None, 'uncalibrated'))},
+ 'recursion': {'all': (metric(0.7, 1.0, 'improved'), metric(0.5, 0.5, 'ok'))}})
+ r['calibration'] = {'quicksort': {'all': {'time': 1.3, 'memory': 0.9}}}
+ text = gate.render_markdown(r)
+ self.assertNotIn('--pr perf-results.json', text)
+ self.assertIn('--reason', text.split('No baseline for')[1].split('')[0])
+
def test_an_incomplete_gate_says_so(self):
text = gate.render_markdown(report({}, error='stale native build'))
self.assertIn('could not complete', text)
@@ -800,6 +809,14 @@ def test_a_stale_overlay_fails_the_job(self):
self.assertEqual(r.returncode, 1, r.stdout + r.stderr)
self.assertIn('pr/7.json is stale', r.stdout)
+ def test_a_calibration_with_a_moved_row_asks_for_a_reason(self):
+ results = {'quicksort': self.row('uncalibrated'), 'recursion': self.row('improved')}
+ r = self.verdict({'platform': 'linux-x64', 'pr': 7, 'labels': {}, 'results': results,
+ 'calibration': {'quicksort': {'all': {'time': 1.0, 'memory': 0.5}}},
+ 'regression': False})
+ self.assertEqual(r.returncode, 1, r.stdout + r.stderr)
+ self.assertIn('--reason', r.stdout)
+
def test_a_judged_run_passes(self):
r = self.verdict({'platform': 'linux-x64', 'results': {'quicksort': self.row('ok')},
'regression': False})