From 7ca5571cbe67a3bfef6f91b48d9e8b6764180bc3 Mon Sep 17 00:00:00 2001 From: Shai Almog <67850168+shai-almog@users.noreply.github.com> Date: Thu, 1 Oct 2026 19:55:38 +0300 Subject: [PATCH 01/11] Performance gate: per-PR baseline overlays, so baselines stop conflicting Every baseline sat in one vm/selfhost/perf-baseline.json. Any branch that met a new runner CPU or moved a benchmark edited it, unrelated rows sat within git's context of each other, and two branches calibrating the same CPU conflicted by construction -- eleven open branches carried edits to it. vm/selfhost/perf-baseline/ now holds policy.json, base/.json and pr/.json. A pull request writes only its own overlay, so the change stays in its diff and no other branch can touch the file. Calibrations of the same CPU combine; a rebaseline records the value it replaces, so two changes moving one benchmark are reported by name instead of as a JSON conflict. A nightly fold moves merged overlays into base/ and refuses to commit unless the resolved baseline is unchanged. The migration is lossless (asserted). An improvement past tolerance now fails the gate until it is rebaselined, and calibration tolerances learn the spread in both directions. The Port Status page gains a ParparVM vs JDK 25 table rendered from these baselines at site-build time; the absolute-time table stays for ports with no JDK arm. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/linux-build-run.yml | 4 + .github/workflows/parparvm-selfhost.yml | 2 +- .github/workflows/parparvm-tests-windows.yml | 8 + .github/workflows/perf-baseline.yml | 111 + .github/workflows/scripts-macos.yml | 4 + .gitignore | 3 + .../assets/css/extended/cn1-port-status.css | 15 + .../website/layouts/_default/port-status.html | 56 + scripts/website/build.sh | 7 + scripts/website/preview.sh | 3 + scripts/website/validate_port_status.mjs | 20 + vm/selfhost/README.md | 37 +- vm/selfhost/calibrate-perf-baseline.py | 240 +- vm/selfhost/ci-perf-gate.sh | 28 +- vm/selfhost/perf-baseline.json | 1930 ----------------- vm/selfhost/perf-baseline/README.md | 11 + .../base/linux-arm64@neoverse-n2.json | 102 + ...x-x64@amd-epyc-7763-64-core-processor.json | 136 ++ ...x-x64@amd-epyc-9v45-96-core-processor.json | 136 ++ ...x-x64@amd-epyc-9v74-80-core-processor.json | 136 ++ .../base/linux-x64@intel-xeon-6973p-c.json | 136 ++ .../linux-x64@intel-xeon-platinum-8370c.json | 136 ++ .../linux-x64@intel-xeon-platinum-8573c.json | 136 ++ .../perf-baseline/base/macos-arm64.json | 124 ++ ...ily-8-model-d49-microsoft-corporation.json | 96 + ...ily-8-model-d84-microsoft-corporation.json | 136 ++ ...@amd64-family-25-model-1-authenticamd.json | 100 + ...amd64-family-25-model-17-authenticamd.json | 136 ++ ...@amd64-family-26-model-2-authenticamd.json | 136 ++ ...tel64-family-6-model-106-genuineintel.json | 136 ++ ...tel64-family-6-model-207-genuineintel.json | 136 ++ vm/selfhost/perf-baseline/policy.json | 10 + vm/selfhost/perf-gate.py | 139 +- vm/selfhost/perf_baseline.py | 581 +++++ vm/selfhost/test_perf_gate.py | 408 +++- 35 files changed, 3395 insertions(+), 2140 deletions(-) create mode 100644 .github/workflows/perf-baseline.yml delete mode 100644 vm/selfhost/perf-baseline.json create mode 100644 vm/selfhost/perf-baseline/README.md create mode 100644 vm/selfhost/perf-baseline/base/linux-arm64@neoverse-n2.json create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-7763-64-core-processor.json create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v45-96-core-processor.json create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@amd-epyc-9v74-80-core-processor.json create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-6973p-c.json create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8370c.json create mode 100644 vm/selfhost/perf-baseline/base/linux-x64@intel-xeon-platinum-8573c.json create mode 100644 vm/selfhost/perf-baseline/base/macos-arm64.json create mode 100644 vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d49-microsoft-corporation.json create mode 100644 vm/selfhost/perf-baseline/base/windows-arm64@armv8-64-bit-family-8-model-d84-microsoft-corporation.json create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-1-authenticamd.json create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@amd64-family-25-model-17-authenticamd.json create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@amd64-family-26-model-2-authenticamd.json create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-106-genuineintel.json create mode 100644 vm/selfhost/perf-baseline/base/windows-x64@intel64-family-6-model-207-genuineintel.json create mode 100644 vm/selfhost/perf-baseline/policy.json create mode 100644 vm/selfhost/perf_baseline.py diff --git a/.github/workflows/linux-build-run.yml b/.github/workflows/linux-build-run.yml index c1e86edfab0..27259c7e8c9 100644 --- a/.github/workflows/linux-build-run.yml +++ b/.github/workflows/linux-build-run.yml @@ -396,6 +396,10 @@ jobs: # results travel with the screenshots. - name: ParparVM vs JDK 25 (performance gate) if: success() || failure() + # Names this pull request's overlay (perf-baseline/pr/.json) in the + # comment. A workflow_dispatch run carries no pull request, so the gate asks gh. + env: + GH_TOKEN: ${{ github.token }} run: | vm/selfhost/ci-perf-gate.sh run linux-${{ matrix.arch }} artifacts/linux-port/raw/perf \ --hello-workload "$GITHUB_WORKSPACE/artifacts/linux-port/raw/translation-record.jsonl" \ diff --git a/.github/workflows/parparvm-selfhost.yml b/.github/workflows/parparvm-selfhost.yml index 6818d5577d0..322b8b0ee70 100644 --- a/.github/workflows/parparvm-selfhost.yml +++ b/.github/workflows/parparvm-selfhost.yml @@ -133,7 +133,7 @@ jobs: # - Keeping it in both places would pay for the expensive -O3 self-host build # twice on any PR carrying the `selfhost` label. # - # The baselines and tolerance live in vm/selfhost/perf-baseline.json, the + # The baselines and tolerance live in vm/selfhost/perf-baseline/, the # measurement in vm/selfhost/perf-gate.py. # Both trees, so a divergence can be inspected rather than guessed at from diff --git a/.github/workflows/parparvm-tests-windows.yml b/.github/workflows/parparvm-tests-windows.yml index 0bd6e65bc04..669fd8a6cdc 100644 --- a/.github/workflows/parparvm-tests-windows.yml +++ b/.github/workflows/parparvm-tests-windows.yml @@ -558,6 +558,10 @@ jobs: # builds only ask javac for -source/-target 1.8 against an explicit bootclasspath. - name: ParparVM vs JDK 25 (performance gate, x64) if: success() || failure() + # Names this pull request's overlay (perf-baseline/pr/.json) in the + # comment. A workflow_dispatch run carries no pull request, so the gate asks gh. + env: + GH_TOKEN: ${{ github.token }} shell: bash run: | export JDK_8_HOME="${JDK_8_HOME:-$JAVA_HOME}" @@ -763,6 +767,10 @@ jobs: # builds only ask javac for -source/-target 1.8 against an explicit bootclasspath. - name: ParparVM vs JDK 25 (performance gate, arm64) if: success() || failure() + # Names this pull request's overlay (perf-baseline/pr/.json) in the + # comment. A workflow_dispatch run carries no pull request, so the gate asks gh. + env: + GH_TOKEN: ${{ github.token }} shell: bash run: | export JDK_8_HOME="${JDK_8_HOME:-$JAVA_HOME}" diff --git a/.github/workflows/perf-baseline.yml b/.github/workflows/perf-baseline.yml new file mode 100644 index 00000000000..5ad54b50169 --- /dev/null +++ b/.github/workflows/perf-baseline.yml @@ -0,0 +1,111 @@ +name: Performance baselines + +# The ParparVM performance gate's baselines live in vm/selfhost/perf-baseline/: +# base/ holds the consolidated rows, and each pull request that changes a baseline +# writes only its own pr/.json, so baseline changes stay in the pull +# request's diff without two branches ever editing the same file +# (vm/selfhost/perf_baseline.py explains the layout). +# +# check every pull request: the overlays resolve against base/ without +# contradicting each other, and the pull request touched only its own +# overlay (base/ is the fold's alone). No paths filter: several of this +# repository's pull requests exceed GitHub's paths-filter diff limit, and a +# filtered check would silently not run on exactly those. +# fold nightly on master: every overlay on master belongs to a merged pull +# request, so it is folded into base/ and deleted, in one commit that +# provably changes no verdict. Then the website is rebuilt, because the +# Port Status page's ParparVM vs JDK 25 table is rendered from these rows. +# A push made with GITHUB_TOKEN triggers no workflow, hence the dispatch. + +on: + pull_request: + branches: + - master + push: + branches: + - master + schedule: + - cron: '20 3 * * *' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.event_name }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + +jobs: + check: + if: github.event_name == 'pull_request' || github.event_name == 'push' + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v6 + with: + # The pull request's base commit has to be present for the diff. + fetch-depth: 0 + + - name: The gate's own unit tests + working-directory: vm/selfhost + run: python3 -m unittest test_perf_gate + + - name: Check the baselines + env: + BASE_SHA: ${{ github.event.pull_request.base.sha }} + PR_NUMBER: ${{ github.event.pull_request.number }} + run: | + if [ -n "${BASE_SHA}" ]; then + python3 vm/selfhost/perf_baseline.py check --base "${BASE_SHA}" --pr "${PR_NUMBER}" + else + python3 vm/selfhost/perf_baseline.py check + fi + + fold: + if: (github.event_name == 'schedule' || github.event_name == 'workflow_dispatch') && github.ref == 'refs/heads/master' + runs-on: ubuntu-24.04 + permissions: + contents: write + actions: write + concurrency: + group: perf-baseline-fold + cancel-in-progress: false + steps: + - uses: actions/checkout@v6 + with: + ref: master + + - name: Fold merged overlays into base/ + id: fold + run: | + set -euo pipefail + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + # A merge landing while this runs makes the push non-fast-forward. Start over + # from the new master rather than rebasing: the fold is cheap, and redoing it + # folds the overlay that just arrived as well. + for attempt in 1 2 3; do + git fetch --quiet origin master + git reset --quiet --hard origin/master + python3 vm/selfhost/perf_baseline.py fold + git add -A vm/selfhost/perf-baseline + if git diff --staged --quiet; then + echo "No overlays to fold." + echo "folded=false" >> "$GITHUB_OUTPUT" + exit 0 + fi + git diff --staged --stat + git commit --quiet -m "ci: fold merged performance baselines into perf-baseline/base" + if git push origin HEAD:master; then + echo "folded=true" >> "$GITHUB_OUTPUT" + exit 0 + fi + echo "master moved during the fold (attempt ${attempt}); retrying." + done + echo "Could not push the fold after 3 attempts." >&2 + exit 1 + + - name: Rebuild the Port Status page + if: steps.fold.outputs.folded == 'true' + env: + GH_TOKEN: ${{ github.token }} + run: gh workflow run website-docs.yml --ref master -f deploy_production=true diff --git a/.github/workflows/scripts-macos.yml b/.github/workflows/scripts-macos.yml index efa6a5794c9..b0d3cb8f7f1 100644 --- a/.github/workflows/scripts-macos.yml +++ b/.github/workflows/scripts-macos.yml @@ -247,6 +247,10 @@ jobs: # got it, and JDK 25 is fetched privately so no later step sees a different JDK. - name: ParparVM vs JDK 25 (performance gate) if: success() || failure() + # Names this pull request's overlay (perf-baseline/pr/.json) in the + # comment. A workflow_dispatch run carries no pull request, so the gate asks gh. + env: + GH_TOKEN: ${{ github.token }} run: | set -u ENV_FILE="${TMPDIR%/}/codenameone-tools/tools/env.sh" diff --git a/.gitignore b/.gitignore index 50a08a7d25c..50246cad5ee 100644 --- a/.gitignore +++ b/.gitignore @@ -193,3 +193,6 @@ cn1-build-hints.json # Concatenated CSS the native-theme build feeds the compiler. Regenerated on # every run; the parts under native-themes//*.css are the source. native-themes/*/target/ + +# Rendered from vm/selfhost/perf-baseline by scripts/website/build.sh; never committed. +docs/website/data/port_status_jdk25.json diff --git a/docs/website/assets/css/extended/cn1-port-status.css b/docs/website/assets/css/extended/cn1-port-status.css index 20e7179f398..8d48938ad1e 100644 --- a/docs/website/assets/css/extended/cn1-port-status.css +++ b/docs/website/assets/css/extended/cn1-port-status.css @@ -490,6 +490,21 @@ min-width: 1320px; } +.cn1-port-status__matrix--jdk25 table { + min-width: 900px; +} + +.cn1-port-status__ratio { + display: block; + white-space: nowrap; +} + +.cn1-port-status__ratio small { + color: var(--secondary); + display: block; + font-size: .68rem; +} + .cn1-port-status__matrix--performance tbody th small { color: var(--secondary); display: block; diff --git a/docs/website/layouts/_default/port-status.html b/docs/website/layouts/_default/port-status.html index e98ae000d65..132241a93ab 100644 --- a/docs/website/layouts/_default/port-status.html +++ b/docs/website/layouts/_default/port-status.html @@ -319,6 +319,62 @@

{{ $support.benchmark.title }}

+ {{- /* Rendered by scripts/website/build.sh from vm/selfhost/perf-baseline: the ratios + every platform build's performance gate holds ParparVM to, so this table and the + gate cannot disagree. Absent from a bare `hugo` run, hence the guard. */ -}} + {{- with site.Data.port_status_jdk25 }} + {{- $jdk := . -}} +
+
+

Gated performance

+

ParparVM against JDK 25

+

+ Each value is ParparVM's time or peak memory divided by HotSpot JDK 25 running the same code on the same CI runner, + as the median of interleaved, paired runs; below 1.00x ParparVM is faster or smaller. These are the baselines every + platform build is gated on: a pull request that moves one past its tolerance, in either direction, fails until the new value is + recorded in the repository. A platform measured on several runner CPU models shows the median across them and, below it, the range. +

+
+
+ + + + + {{- range $jdk.platforms }}{{- end }} + + + + {{- range $jdk.benchmarks }} + {{- $benchmark := . -}} + + + {{- range $jdk.platforms }} + {{- $platform := . -}} + {{- $cell := index .benchmarks $benchmark.id -}} + + {{- end }} + + {{- end }} + +
Workload{{ .name }}
{{ .name }}

{{ .description }}

Time and RAM, ParparVM / JDK 25
+ {{- with $cell -}} + {{- range $metric := slice "time" "memory" -}} + {{- $m := index $cell $metric -}} + {{- $lo := printf "%.2f" (float $m.min) -}} + {{- $hi := printf "%.2f" (float $m.max) -}} + {{ if eq $metric "time" }}Time{{ else }}RAM{{ end }} {{ printf "%.2fx" (float $m.median) }} + {{- if ne $lo $hi }}{{ $lo }}-{{ $hi }}x{{ end -}} + + {{- end -}} + {{- else -}}Not measured{{- end -}} +
+
+

+ Performance gate baselines +

+
+ {{- end }} +