diff --git a/.github/workflows/benchmark-clawbench.yml b/.github/workflows/benchmark-clawbench.yml index 9f035ab4..a701e6df 100644 --- a/.github/workflows/benchmark-clawbench.yml +++ b/.github/workflows/benchmark-clawbench.yml @@ -225,6 +225,7 @@ jobs: CODEX_BENCHMARK_VERSION: "0.120.0" CLAUDE_BENCHMARK_MODEL: claude-sonnet-5 CLAUDE_BENCHMARK_VERSION: "2.1.238" + CLAWBENCH_REF: 187cd252bc60af8ac3a2c98a87c9316e49a5ac75 HARBOR_N_CONCURRENT: ${{ needs.resolve.outputs.concurrency }} BENCHMARK_PR_NUMBER: ${{ needs.resolve.outputs.pr_number }} BENCHMARK_HEAD_SHA: ${{ needs.resolve.outputs.head_sha }} @@ -284,7 +285,7 @@ jobs: uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 with: repository: kernel/ClawBench - ref: c7feaa2435ca8115c0762c44e13885fe5adf3e98 + ref: ${{ env.CLAWBENCH_REF }} path: clawbench persist-credentials: false @@ -326,7 +327,6 @@ jobs: shell: bash env: CLAWBENCH_REPO: ${{ github.workspace }}/clawbench - CLAWBENCH_REF: c7feaa2435ca8115c0762c44e13885fe5adf3e98 HARBOR_BENCHMARK_TIMEOUT: 4h BENCHMARK_AGENT: ${{ needs.resolve.outputs.agent }} BENCHMARK_TASK: ${{ needs.resolve.outputs.task }} diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index 08566136..d52eff34 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -18,7 +18,7 @@ The image records the current Git SHA, and the generated task records the ClawBe - `uv`, Harbor 0.21.0, and `harbor-hypeman` 0.1.2 - Hypeman CLI credentials -- a ClawBench checkout containing pinned commit `c7feaa2` +- a [`kernel/ClawBench`](https://github.com/kernel/ClawBench) checkout containing pinned commit `187cd252bc60af8ac3a2c98a87c9316e49a5ac75` - `KERNEL_MCP_BENCHMARK_API_KEY` scoped to an isolated evaluation project; its credential scope is the project source of truth - `PURELY_MAIL_API_KEY` and `PURELY_MAIL_DOMAIN` for ClawBench account tasks - `OPENAI_API_KEY` for Codex, or Anthropic credentials for Claude Code diff --git a/benchmarks/harbor/clawbench/run.sh b/benchmarks/harbor/clawbench/run.sh index 04e84175..1fb32a52 100755 --- a/benchmarks/harbor/clawbench/run.sh +++ b/benchmarks/harbor/clawbench/run.sh @@ -20,7 +20,7 @@ source_root=${KERNEL_MCP_BENCHMARK_SOURCE_ROOT:-$harness_root} benchmark_dir="$harness_root/benchmarks/harbor" image_env="$source_root/benchmarks/harbor/.image.env" clawbench_repo=${CLAWBENCH_REPO:-$harness_root/../ClawBench} -clawbench_ref=${CLAWBENCH_REF:-c7feaa2435ca8115c0762c44e13885fe5adf3e98} +clawbench_ref=${CLAWBENCH_REF:-187cd252bc60af8ac3a2c98a87c9316e49a5ac75} [[ -f "$image_env" ]] || { echo "Missing $image_env; run benchmarks/harbor/build-image.sh first" >&2 diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 1182ea84..78f93531 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -913,14 +913,30 @@ describe("benchmark workflow hardening", () => { join(process.cwd(), ".github/workflows/benchmark-clawbench.yml"), "utf8", ); + const runner = readFileSync( + join(process.cwd(), "benchmarks/harbor/clawbench/run.sh"), + "utf8", + ); + const readme = readFileSync( + join(process.cwd(), "benchmarks/harbor/README.md"), + "utf8", + ); + const refMatch = runner.match( + /clawbench_ref=\$\{CLAWBENCH_REF:-([0-9a-f]{40})\}/, + ); + if (!refMatch) throw new Error("runner is missing the ClawBench pin"); + const clawbenchRef = refMatch[1]; + expect(workflow).toContain("github.rest.repos.compareCommits"); expect(workflow).not.toContain("baseSha = pull.base.sha"); expect(workflow).toContain('HARBOR_VERSION: "0.21.0"'); expect(workflow).toContain('HARBOR_HYPEMAN_VERSION: "0.1.2"'); expect(workflow).toContain('CODEX_BENCHMARK_VERSION: "0.120.0"'); - expect( - workflow.match(/c7feaa2435ca8115c0762c44e13885fe5adf3e98/g), - ).toHaveLength(2); + expect(workflow.match(new RegExp(clawbenchRef, "g"))).toHaveLength(1); + expect(workflow).toContain(`CLAWBENCH_REF: ${clawbenchRef}`); + expect(workflow).toContain("ref: ${{ env.CLAWBENCH_REF }}"); + expect(readme).toContain("https://github.com/kernel/ClawBench"); + expect(readme).toContain(clawbenchRef); expect(workflow).toContain("issues: write\n pull-requests: write"); expect(workflow).not.toContain( "KERNEL_PROJECT: ${{ vars.KERNEL_PROJECT }}", @@ -940,10 +956,6 @@ describe("benchmark workflow hardening", () => { '"$GITHUB_WORKSPACE/harness/benchmarks/harbor/clawbench/run.sh"', ); - const runner = readFileSync( - join(process.cwd(), "benchmarks/harbor/clawbench/run.sh"), - "utf8", - ); expect(runner).toContain( "source_root=${KERNEL_MCP_BENCHMARK_SOURCE_ROOT:-$harness_root}", ); @@ -1012,7 +1024,6 @@ describe("benchmark workflow hardening", () => { ); expect(dockerignore.split("\n")).toContain("*.pem"); expect(runner).not.toContain("KERNEL_PROJECT"); - expect(runner).toContain("c7feaa2435ca8115c0762c44e13885fe5adf3e98"); expect(runner).toContain('"${KERNEL_API_BASE_URL%/}/auth/context"'); expect(runner).toContain('bun "$benchmark_dir/verify-project-scope.ts"'); expect(taskPreparer).not.toContain("KERNEL_PROJECT");