From ea0607c48fb92d7d2e9481b4aa69132a97943b7d Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Tue, 8 Sep 2026 20:17:48 +0000 Subject: [PATCH 1/3] Update the pinned ClawBench revision --- .github/workflows/benchmark-clawbench.yml | 4 ++-- benchmarks/harbor/clawbench/run.sh | 2 +- benchmarks/harbor/results.test.ts | 4 ++-- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/.github/workflows/benchmark-clawbench.yml b/.github/workflows/benchmark-clawbench.yml index 9f035ab4..478340ef 100644 --- a/.github/workflows/benchmark-clawbench.yml +++ b/.github/workflows/benchmark-clawbench.yml @@ -284,7 +284,7 @@ jobs: uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 with: repository: kernel/ClawBench - ref: c7feaa2435ca8115c0762c44e13885fe5adf3e98 + ref: 9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df path: clawbench persist-credentials: false @@ -326,7 +326,7 @@ jobs: shell: bash env: CLAWBENCH_REPO: ${{ github.workspace }}/clawbench - CLAWBENCH_REF: c7feaa2435ca8115c0762c44e13885fe5adf3e98 + CLAWBENCH_REF: 9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df HARBOR_BENCHMARK_TIMEOUT: 4h BENCHMARK_AGENT: ${{ needs.resolve.outputs.agent }} BENCHMARK_TASK: ${{ needs.resolve.outputs.task }} diff --git a/benchmarks/harbor/clawbench/run.sh b/benchmarks/harbor/clawbench/run.sh index 1feda7e2..03f9e580 100755 --- a/benchmarks/harbor/clawbench/run.sh +++ b/benchmarks/harbor/clawbench/run.sh @@ -20,7 +20,7 @@ source_root=${KERNEL_MCP_BENCHMARK_SOURCE_ROOT:-$harness_root} benchmark_dir="$harness_root/benchmarks/harbor" image_env="$source_root/benchmarks/harbor/.image.env" clawbench_repo=${CLAWBENCH_REPO:-$harness_root/../ClawBench} -clawbench_ref=${CLAWBENCH_REF:-c7feaa2435ca8115c0762c44e13885fe5adf3e98} +clawbench_ref=${CLAWBENCH_REF:-9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df} [[ -f "$image_env" ]] || { echo "Missing $image_env; run benchmarks/harbor/build-image.sh first" >&2 diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index a0774091..3766728d 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -441,7 +441,7 @@ describe("benchmark workflow hardening", () => { expect(workflow).toContain('HARBOR_HYPEMAN_VERSION: "0.1.2"'); expect(workflow).toContain('CODEX_BENCHMARK_VERSION: "0.120.0"'); expect( - workflow.match(/c7feaa2435ca8115c0762c44e13885fe5adf3e98/g), + workflow.match(/9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df/g), ).toHaveLength(2); expect(workflow).toContain("issues: write\n pull-requests: write"); expect(workflow).not.toContain( @@ -531,7 +531,7 @@ describe("benchmark workflow hardening", () => { ); expect(dockerignore.split("\n")).toContain("*.pem"); expect(runner).not.toContain("KERNEL_PROJECT"); - expect(runner).toContain("c7feaa2435ca8115c0762c44e13885fe5adf3e98"); + expect(runner).toContain("9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df"); expect(runner).toContain('"${KERNEL_API_BASE_URL%/}/auth/context"'); expect(runner).toContain('bun "$benchmark_dir/verify-project-scope.ts"'); expect(taskPreparer).not.toContain("KERNEL_PROJECT"); From f704ee668fb0c3c14405478486a31fd90f328be2 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Thu, 1 Oct 2026 12:54:38 +0000 Subject: [PATCH 2/3] Pin ClawBench to latest upstream --- .github/workflows/benchmark-clawbench.yml | 4 ++-- benchmarks/harbor/README.md | 2 +- benchmarks/harbor/clawbench/run.sh | 2 +- benchmarks/harbor/results.test.ts | 4 ++-- 4 files changed, 6 insertions(+), 6 deletions(-) diff --git a/.github/workflows/benchmark-clawbench.yml b/.github/workflows/benchmark-clawbench.yml index 478340ef..a430abf3 100644 --- a/.github/workflows/benchmark-clawbench.yml +++ b/.github/workflows/benchmark-clawbench.yml @@ -284,7 +284,7 @@ jobs: uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 with: repository: kernel/ClawBench - ref: 9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df + ref: 187cd252bc60af8ac3a2c98a87c9316e49a5ac75 path: clawbench persist-credentials: false @@ -326,7 +326,7 @@ jobs: shell: bash env: CLAWBENCH_REPO: ${{ github.workspace }}/clawbench - CLAWBENCH_REF: 9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df + CLAWBENCH_REF: 187cd252bc60af8ac3a2c98a87c9316e49a5ac75 HARBOR_BENCHMARK_TIMEOUT: 4h BENCHMARK_AGENT: ${{ needs.resolve.outputs.agent }} BENCHMARK_TASK: ${{ needs.resolve.outputs.task }} diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index 7b7b2ce2..fdc7c205 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -18,7 +18,7 @@ The image records the current Git SHA, and the generated task records the ClawBe - `uv`, Harbor 0.21.0, and `harbor-hypeman` 0.1.2 - Hypeman CLI credentials -- a ClawBench checkout containing pinned commit `c7feaa2` +- a ClawBench checkout containing pinned commit `187cd25` - `KERNEL_MCP_BENCHMARK_API_KEY` scoped to an isolated evaluation project; its credential scope is the project source of truth - `PURELY_MAIL_API_KEY` and `PURELY_MAIL_DOMAIN` for ClawBench account tasks - `OPENAI_API_KEY` for Codex, or Anthropic credentials for Claude Code diff --git a/benchmarks/harbor/clawbench/run.sh b/benchmarks/harbor/clawbench/run.sh index 03f9e580..83e397fc 100755 --- a/benchmarks/harbor/clawbench/run.sh +++ b/benchmarks/harbor/clawbench/run.sh @@ -20,7 +20,7 @@ source_root=${KERNEL_MCP_BENCHMARK_SOURCE_ROOT:-$harness_root} benchmark_dir="$harness_root/benchmarks/harbor" image_env="$source_root/benchmarks/harbor/.image.env" clawbench_repo=${CLAWBENCH_REPO:-$harness_root/../ClawBench} -clawbench_ref=${CLAWBENCH_REF:-9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df} +clawbench_ref=${CLAWBENCH_REF:-187cd252bc60af8ac3a2c98a87c9316e49a5ac75} [[ -f "$image_env" ]] || { echo "Missing $image_env; run benchmarks/harbor/build-image.sh first" >&2 diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 3766728d..3fd67b16 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -441,7 +441,7 @@ describe("benchmark workflow hardening", () => { expect(workflow).toContain('HARBOR_HYPEMAN_VERSION: "0.1.2"'); expect(workflow).toContain('CODEX_BENCHMARK_VERSION: "0.120.0"'); expect( - workflow.match(/9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df/g), + workflow.match(/187cd252bc60af8ac3a2c98a87c9316e49a5ac75/g), ).toHaveLength(2); expect(workflow).toContain("issues: write\n pull-requests: write"); expect(workflow).not.toContain( @@ -531,7 +531,7 @@ describe("benchmark workflow hardening", () => { ); expect(dockerignore.split("\n")).toContain("*.pem"); expect(runner).not.toContain("KERNEL_PROJECT"); - expect(runner).toContain("9dd9d442c8ec720ba4d6d4c95f2bbc0ff2a2d0df"); + expect(runner).toContain("187cd252bc60af8ac3a2c98a87c9316e49a5ac75"); expect(runner).toContain('"${KERNEL_API_BASE_URL%/}/auth/context"'); expect(runner).toContain('bun "$benchmark_dir/verify-project-scope.ts"'); expect(taskPreparer).not.toContain("KERNEL_PROJECT"); From eae1b57dcc23e56b70989b19416ee55c3edb5741 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Thu, 1 Oct 2026 13:17:41 +0000 Subject: [PATCH 3/3] Centralize the ClawBench pin --- .github/workflows/benchmark-clawbench.yml | 4 ++-- benchmarks/harbor/README.md | 2 +- benchmarks/harbor/results.test.ts | 27 ++++++++++++++++------- 3 files changed, 22 insertions(+), 11 deletions(-) diff --git a/.github/workflows/benchmark-clawbench.yml b/.github/workflows/benchmark-clawbench.yml index a430abf3..a701e6df 100644 --- a/.github/workflows/benchmark-clawbench.yml +++ b/.github/workflows/benchmark-clawbench.yml @@ -225,6 +225,7 @@ jobs: CODEX_BENCHMARK_VERSION: "0.120.0" CLAUDE_BENCHMARK_MODEL: claude-sonnet-5 CLAUDE_BENCHMARK_VERSION: "2.1.238" + CLAWBENCH_REF: 187cd252bc60af8ac3a2c98a87c9316e49a5ac75 HARBOR_N_CONCURRENT: ${{ needs.resolve.outputs.concurrency }} BENCHMARK_PR_NUMBER: ${{ needs.resolve.outputs.pr_number }} BENCHMARK_HEAD_SHA: ${{ needs.resolve.outputs.head_sha }} @@ -284,7 +285,7 @@ jobs: uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 with: repository: kernel/ClawBench - ref: 187cd252bc60af8ac3a2c98a87c9316e49a5ac75 + ref: ${{ env.CLAWBENCH_REF }} path: clawbench persist-credentials: false @@ -326,7 +327,6 @@ jobs: shell: bash env: CLAWBENCH_REPO: ${{ github.workspace }}/clawbench - CLAWBENCH_REF: 187cd252bc60af8ac3a2c98a87c9316e49a5ac75 HARBOR_BENCHMARK_TIMEOUT: 4h BENCHMARK_AGENT: ${{ needs.resolve.outputs.agent }} BENCHMARK_TASK: ${{ needs.resolve.outputs.task }} diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index fdc7c205..2a352f87 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -18,7 +18,7 @@ The image records the current Git SHA, and the generated task records the ClawBe - `uv`, Harbor 0.21.0, and `harbor-hypeman` 0.1.2 - Hypeman CLI credentials -- a ClawBench checkout containing pinned commit `187cd25` +- a [`kernel/ClawBench`](https://github.com/kernel/ClawBench) checkout containing pinned commit `187cd252bc60af8ac3a2c98a87c9316e49a5ac75` - `KERNEL_MCP_BENCHMARK_API_KEY` scoped to an isolated evaluation project; its credential scope is the project source of truth - `PURELY_MAIL_API_KEY` and `PURELY_MAIL_DOMAIN` for ClawBench account tasks - `OPENAI_API_KEY` for Codex, or Anthropic credentials for Claude Code diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 3fd67b16..01478f4f 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -435,14 +435,30 @@ describe("benchmark workflow hardening", () => { join(process.cwd(), ".github/workflows/benchmark-clawbench.yml"), "utf8", ); + const runner = readFileSync( + join(process.cwd(), "benchmarks/harbor/clawbench/run.sh"), + "utf8", + ); + const readme = readFileSync( + join(process.cwd(), "benchmarks/harbor/README.md"), + "utf8", + ); + const refMatch = runner.match( + /clawbench_ref=\$\{CLAWBENCH_REF:-([0-9a-f]{40})\}/, + ); + if (!refMatch) throw new Error("runner is missing the ClawBench pin"); + const clawbenchRef = refMatch[1]; + expect(workflow).toContain("github.rest.repos.compareCommits"); expect(workflow).not.toContain("baseSha = pull.base.sha"); expect(workflow).toContain('HARBOR_VERSION: "0.21.0"'); expect(workflow).toContain('HARBOR_HYPEMAN_VERSION: "0.1.2"'); expect(workflow).toContain('CODEX_BENCHMARK_VERSION: "0.120.0"'); - expect( - workflow.match(/187cd252bc60af8ac3a2c98a87c9316e49a5ac75/g), - ).toHaveLength(2); + expect(workflow.match(new RegExp(clawbenchRef, "g"))).toHaveLength(1); + expect(workflow).toContain(`CLAWBENCH_REF: ${clawbenchRef}`); + expect(workflow).toContain("ref: ${{ env.CLAWBENCH_REF }}"); + expect(readme).toContain("https://github.com/kernel/ClawBench"); + expect(readme).toContain(clawbenchRef); expect(workflow).toContain("issues: write\n pull-requests: write"); expect(workflow).not.toContain( "KERNEL_PROJECT: ${{ vars.KERNEL_PROJECT }}", @@ -462,10 +478,6 @@ describe("benchmark workflow hardening", () => { '"$GITHUB_WORKSPACE/harness/benchmarks/harbor/clawbench/run.sh"', ); - const runner = readFileSync( - join(process.cwd(), "benchmarks/harbor/clawbench/run.sh"), - "utf8", - ); expect(runner).toContain( "source_root=${KERNEL_MCP_BENCHMARK_SOURCE_ROOT:-$harness_root}", ); @@ -531,7 +543,6 @@ describe("benchmark workflow hardening", () => { ); expect(dockerignore.split("\n")).toContain("*.pem"); expect(runner).not.toContain("KERNEL_PROJECT"); - expect(runner).toContain("187cd252bc60af8ac3a2c98a87c9316e49a5ac75"); expect(runner).toContain('"${KERNEL_API_BASE_URL%/}/auth/context"'); expect(runner).toContain('bun "$benchmark_dir/verify-project-scope.ts"'); expect(taskPreparer).not.toContain("KERNEL_PROJECT");