Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions .github/workflows/benchmark-clawbench.yml
Original file line number Diff line number Diff line change
Expand Up @@ -225,6 +225,7 @@ jobs:
CODEX_BENCHMARK_VERSION: "0.120.0"
CLAUDE_BENCHMARK_MODEL: claude-sonnet-5
CLAUDE_BENCHMARK_VERSION: "2.1.238"
CLAWBENCH_REF: 187cd252bc60af8ac3a2c98a87c9316e49a5ac75
HARBOR_N_CONCURRENT: ${{ needs.resolve.outputs.concurrency }}
BENCHMARK_PR_NUMBER: ${{ needs.resolve.outputs.pr_number }}
BENCHMARK_HEAD_SHA: ${{ needs.resolve.outputs.head_sha }}
Expand Down Expand Up @@ -284,7 +285,7 @@ jobs:
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1
with:
repository: kernel/ClawBench
ref: c7feaa2435ca8115c0762c44e13885fe5adf3e98
ref: ${{ env.CLAWBENCH_REF }}
path: clawbench
persist-credentials: false

Expand Down Expand Up @@ -326,7 +327,6 @@ jobs:
shell: bash
env:
CLAWBENCH_REPO: ${{ github.workspace }}/clawbench
CLAWBENCH_REF: c7feaa2435ca8115c0762c44e13885fe5adf3e98
HARBOR_BENCHMARK_TIMEOUT: 4h
BENCHMARK_AGENT: ${{ needs.resolve.outputs.agent }}
BENCHMARK_TASK: ${{ needs.resolve.outputs.task }}
Expand Down
2 changes: 1 addition & 1 deletion benchmarks/harbor/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,7 @@ The image records the current Git SHA, and the generated task records the ClawBe

- `uv`, Harbor 0.21.0, and `harbor-hypeman` 0.1.2
- Hypeman CLI credentials
- a ClawBench checkout containing pinned commit `c7feaa2`
- a [`kernel/ClawBench`](https://github.com/kernel/ClawBench) checkout containing pinned commit `187cd252bc60af8ac3a2c98a87c9316e49a5ac75`
- `KERNEL_MCP_BENCHMARK_API_KEY` scoped to an isolated evaluation project; its credential scope is the project source of truth
- `PURELY_MAIL_API_KEY` and `PURELY_MAIL_DOMAIN` for ClawBench account tasks
- `OPENAI_API_KEY` for Codex, or Anthropic credentials for Claude Code
Expand Down
2 changes: 1 addition & 1 deletion benchmarks/harbor/clawbench/run.sh
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ source_root=${KERNEL_MCP_BENCHMARK_SOURCE_ROOT:-$harness_root}
benchmark_dir="$harness_root/benchmarks/harbor"
image_env="$source_root/benchmarks/harbor/.image.env"
clawbench_repo=${CLAWBENCH_REPO:-$harness_root/../ClawBench}
clawbench_ref=${CLAWBENCH_REF:-c7feaa2435ca8115c0762c44e13885fe5adf3e98}
clawbench_ref=${CLAWBENCH_REF:-187cd252bc60af8ac3a2c98a87c9316e49a5ac75}

[[ -f "$image_env" ]] || {
echo "Missing $image_env; run benchmarks/harbor/build-image.sh first" >&2
Expand Down
27 changes: 19 additions & 8 deletions benchmarks/harbor/results.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -913,14 +913,30 @@ describe("benchmark workflow hardening", () => {
join(process.cwd(), ".github/workflows/benchmark-clawbench.yml"),
"utf8",
);
const runner = readFileSync(
join(process.cwd(), "benchmarks/harbor/clawbench/run.sh"),
"utf8",
);
const readme = readFileSync(
join(process.cwd(), "benchmarks/harbor/README.md"),
"utf8",
);
const refMatch = runner.match(
/clawbench_ref=\$\{CLAWBENCH_REF:-([0-9a-f]{40})\}/,
);
if (!refMatch) throw new Error("runner is missing the ClawBench pin");
const clawbenchRef = refMatch[1];

expect(workflow).toContain("github.rest.repos.compareCommits");
expect(workflow).not.toContain("baseSha = pull.base.sha");
expect(workflow).toContain('HARBOR_VERSION: "0.21.0"');
expect(workflow).toContain('HARBOR_HYPEMAN_VERSION: "0.1.2"');
expect(workflow).toContain('CODEX_BENCHMARK_VERSION: "0.120.0"');
expect(
workflow.match(/c7feaa2435ca8115c0762c44e13885fe5adf3e98/g),
).toHaveLength(2);
expect(workflow.match(new RegExp(clawbenchRef, "g"))).toHaveLength(1);
expect(workflow).toContain(`CLAWBENCH_REF: ${clawbenchRef}`);
expect(workflow).toContain("ref: ${{ env.CLAWBENCH_REF }}");
expect(readme).toContain("https://github.com/kernel/ClawBench");
expect(readme).toContain(clawbenchRef);
expect(workflow).toContain("issues: write\n pull-requests: write");
expect(workflow).not.toContain(
"KERNEL_PROJECT: ${{ vars.KERNEL_PROJECT }}",
Expand All @@ -940,10 +956,6 @@ describe("benchmark workflow hardening", () => {
'"$GITHUB_WORKSPACE/harness/benchmarks/harbor/clawbench/run.sh"',
);

const runner = readFileSync(
join(process.cwd(), "benchmarks/harbor/clawbench/run.sh"),
"utf8",
);
expect(runner).toContain(
"source_root=${KERNEL_MCP_BENCHMARK_SOURCE_ROOT:-$harness_root}",
);
Expand Down Expand Up @@ -1012,7 +1024,6 @@ describe("benchmark workflow hardening", () => {
);
expect(dockerignore.split("\n")).toContain("*.pem");
expect(runner).not.toContain("KERNEL_PROJECT");
expect(runner).toContain("c7feaa2435ca8115c0762c44e13885fe5adf3e98");
expect(runner).toContain('"${KERNEL_API_BASE_URL%/}/auth/context"');
expect(runner).toContain('bun "$benchmark_dir/verify-project-scope.ts"');
expect(taskPreparer).not.toContain("KERNEL_PROJECT");
Expand Down
Loading