diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index 7b7b2ce2..08566136 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -57,7 +57,9 @@ Codex defaults to version `0.120.0` with `gpt-5.6-luna`. Claude Code defaults to Single-task runs have a 40-minute wall-clock limit. Full-suite runs default to six hours. Set `HARBOR_BENCHMARK_TIMEOUT` to override either limit. Set `HARBOR_JOBS_DIR` to choose where Harbor writes results. -The runner retries a whole isolated trial up to five times for transient Hypeman connection, timeout, and exec-stream failures. Set `HARBOR_MAX_RETRIES` to override that limit. Per-request SDK retries remain disabled because transparently retrying instance or image creation can duplicate a request whose first response was lost. +Before creating the Harbor dataset, the runner creates and deletes one disposable PurelyMail account. API errors stop the run before trials begin, so missing email accounts cannot silently become task failures. + +The runner retries a whole isolated trial up to five times for transient Hypeman connection, timeout, exec-stream, and agent setup failures. Harbor step `setup.sh` failures are not retried; the PurelyMail preflight catches the known external setup dependency before trials begin. Set `HARBOR_MAX_RETRIES` to override the trial retry limit. Per-request SDK retries remain disabled because transparently retrying instance or image creation can duplicate a request whose first response was lost. ## GitHub Actions @@ -102,6 +104,8 @@ BRAINTRUST_PROJECT=kernel-mcp-server-benchmarks \ --arm baseline=/path/to/baseline-job ``` -The experiment name and deterministic row/span IDs make it safe to publish the same job directories again. Re-publication replaces the rows and refreshes experiment metadata. Rows contain task identity, numeric rewards, provenance, bounded errors, timing, token, call, and cost metrics. ATIF agent/tool activity is attached as child spans after secret redaction. Task instructions, ground truth, browser session URLs, and recordings are not placed on experiment rows or public pull-request comments. +The experiment name and deterministic row/span IDs make it safe to publish the same job directories again. Re-publication replaces the rows and refreshes experiment metadata. Rows contain task identity, the redacted instruction, numeric rewards, provenance, bounded errors, and trial timing. Setup, browser lifetime, execution, verification, and finalization are separate timeline spans; the browser span records its timeout and deletion status without its identifiers. ATIF turns include their preceding context, structured tool calls, cached-token metrics, and inferred turn intervals. Browser tool spans record whether the supplied session ID matched the trial's expected session without publishing either ID. Tool durations remain unspecified because ATIF does not record them. + +The publisher redacts typed form values, configured secrets, credentials, email addresses, browser session/replay IDs, private-info file contents, and provider URLs. It validates the final payload and aborts before creating an experiment if sensitive content remains. Ground truth, recordings, and raw Harbor jobs are never published. Reports suppress comparison deltas when either arm has an infrastructure failure or ungraded trial. The workflow fails unless every intended trial is graded, while still retaining the incomplete report for diagnosis. diff --git a/benchmarks/harbor/clawbench/run.sh b/benchmarks/harbor/clawbench/run.sh index 1feda7e2..04e84175 100755 --- a/benchmarks/harbor/clawbench/run.sh +++ b/benchmarks/harbor/clawbench/run.sh @@ -48,6 +48,8 @@ set +a : "${PURELY_MAIL_API_KEY:?PURELY_MAIL_API_KEY is required}" : "${PURELY_MAIL_DOMAIN:?PURELY_MAIL_DOMAIN is required}" +bun "$benchmark_dir/verify-purelymail.ts" + case "$agent" in claude-code) if [[ -z "${ANTHROPIC_API_KEY:-}" && -z "${ANTHROPIC_AUTH_TOKEN:-}" && -z "${CLAUDE_CODE_OAUTH_TOKEN:-}" ]]; then @@ -169,5 +171,7 @@ timeout --signal=INT --kill-after=30s "${HARBOR_BENCHMARK_TIMEOUT:-$default_time --retry-include InternalServerError \ --retry-include ConnectionRefusedError \ --retry-include ExecProtocolError \ + --retry-include AgentSetupTimeoutError \ + --retry-include RuntimeError \ --delete \ --yes diff --git a/benchmarks/harbor/publish-braintrust.ts b/benchmarks/harbor/publish-braintrust.ts index c6fa1c1b..2f7d7347 100644 --- a/benchmarks/harbor/publish-braintrust.ts +++ b/benchmarks/harbor/publish-braintrust.ts @@ -4,13 +4,23 @@ import { mkdirSync, readFileSync, writeFileSync } from "node:fs"; import { dirname } from "node:path"; import { type BenchmarkArm, + type BenchmarkPhase, type BenchmarkTrial, parseArmSpec, readBenchmarkArm, selectPrimaryReward, summarizeArm, } from "./results"; -import { redactString, redactValue } from "./redact"; +import { + assertSafeToPublish, + collectPrivateInfoValues, + collectSensitiveValues, + privateInfoRead, + redactString, + redactValue, + redactValueWithSecrets, + REDACTED_PRIVATE_INFO, +} from "./redact"; interface CliOptions { arms: string[]; @@ -41,7 +51,7 @@ interface BraintrustEvent { span_id: string; root_span_id: string; span_parents: string[]; - span_attributes: { name: string; type: "eval" | "llm" | "tool" }; + span_attributes: { name: string; type: "eval" | "llm" | "tool" | "task" }; created?: string; input?: unknown; output?: unknown; @@ -123,20 +133,33 @@ function trajectorySteps(trial: BenchmarkTrial): AtifStep[] { } function metricRecord(trial: BenchmarkTrial): Record { + const promptTokens = trial.metrics.inputTokens; + const completionTokens = trial.metrics.outputTokens; return Object.fromEntries( Object.entries({ start: trial.metrics.start, end: trial.metrics.end, - input_tokens: trial.metrics.inputTokens, - cached_tokens: trial.metrics.cacheTokens, - output_tokens: trial.metrics.outputTokens, - cost_usd: trial.metrics.costUsd, duration_ms: trial.metrics.durationMs, tool_calls: trial.metrics.toolCalls, + prompt_tokens: promptTokens, + prompt_cached_tokens: trial.metrics.cacheTokens, + completion_tokens: completionTokens, + tokens: + promptTokens === undefined || completionTokens === undefined + ? undefined + : promptTokens + completionTokens, + estimated_cost: trial.metrics.costUsd, }).filter((entry): entry is [string, number] => entry[1] !== undefined), ); } +function taskInstruction(trial: BenchmarkTrial): unknown { + const userSteps = trajectorySteps(trial).filter( + (step) => step.source === "user" && step.message !== undefined, + ); + return redactValue(userSteps.at(-1)?.message); +} + function trialMetadata(trial: BenchmarkTrial): Record { return { trialName: trial.trialName, @@ -161,39 +184,243 @@ function trialMetadata(trial: BenchmarkTrial): Record { }; } +function timestampSeconds(value?: string): number | undefined { + if (!value) return undefined; + const millis = Date.parse(value); + return Number.isFinite(millis) ? millis / 1000 : undefined; +} + +function timestampIso(value?: number): string | undefined { + return value === undefined ? undefined : new Date(value * 1000).toISOString(); +} + +function stepContext( + step: AtifStep, + sensitiveValues: string[], +): Record { + const calls = step.tool_calls ?? []; + return { + source: step.source, + message: redactValueWithSecrets(step.message, sensitiveValues), + toolCalls: calls.map((call) => ({ + name: call.function_name ?? "tool", + arguments: redactValueWithSecrets(call.arguments, sensitiveValues), + })), + observations: (step.observation?.results ?? []).map((result) => { + const call = calls.find( + (candidate) => candidate.tool_call_id === result.source_call_id, + ); + const toolName = call?.function_name ?? "tool"; + return { + source_call_id: result.source_call_id, + content: + call && privateInfoRead(toolName, call.arguments) + ? REDACTED_PRIVATE_INFO + : redactValueWithSecrets(result.content, sensitiveValues), + }; + }), + }; +} + +function llmInput( + steps: AtifStep[], + stepIndex: number, + sensitiveValues: string[], +): unknown { + const prior = steps.slice(0, stepIndex); + let previousAgent = -1; + for (let index = prior.length - 1; index >= 0; index -= 1) { + if (prior[index].source === "agent") { + previousAgent = index; + break; + } + } + const context = previousAgent === -1 ? prior : prior.slice(previousAgent); + return context.map((step) => stepContext(step, sensitiveValues)); +} + +function llmOutput(step: AtifStep, sensitiveValues: string[]): unknown { + if ((step.tool_calls ?? []).length === 0) { + return redactValueWithSecrets(step.message, sensitiveValues); + } + return { + message: redactValueWithSecrets(step.message, sensitiveValues), + toolCalls: (step.tool_calls ?? []).map((call) => ({ + name: call.function_name ?? "tool", + arguments: redactValueWithSecrets(call.arguments, sensitiveValues), + })), + }; +} + +function privateInfoValues(steps: AtifStep[]): string[] { + const values = new Set(); + for (const step of steps) { + for (const call of step.tool_calls ?? []) { + const toolName = call.function_name ?? "tool"; + if (!privateInfoRead(toolName, call.arguments)) continue; + const observation = step.observation?.results?.find( + (result) => result.source_call_id === call.tool_call_id, + ); + if (typeof observation?.content !== "string") continue; + let privateInfo: unknown = observation.content; + try { + privateInfo = JSON.parse(observation.content); + } catch { + // The collector also handles key/value pairs in non-JSON output. + } + for (const value of collectPrivateInfoValues(privateInfo)) { + values.add(value); + } + } + } + return [...values]; +} + +function phaseEvent( + rowId: string, + name: string, + phase: BenchmarkPhase, + metadata: Record = {}, +): BraintrustEvent | undefined { + if (phase.start === undefined || phase.end === undefined) return undefined; + const id = uuidV5(`${rowId}:phase:${name}`); + return { + id, + span_id: id, + root_span_id: rowId, + span_parents: [rowId], + span_attributes: { name, type: "task" }, + created: timestampIso(phase.start), + metadata: { phase: name, ...metadata }, + metrics: { + start: phase.start, + end: phase.end, + ...(phase.durationMs === undefined + ? {} + : { duration_ms: phase.durationMs }), + }, + _is_merge: false, + }; +} + +function phaseEvents(trial: BenchmarkTrial, rowId: string): BraintrustEvent[] { + const events: BraintrustEvent[] = []; + for (const [name, phase] of Object.entries({ + environment_setup: trial.phases.environmentSetup, + agent_setup: trial.phases.agentSetup, + agent_execution: trial.phases.agentExecution, + verifier: trial.phases.verifier, + })) { + if (!phase) continue; + const event = phaseEvent(rowId, name, phase); + if (event) events.push(event); + } + + const stepSetupStart = trial.phases.agentSetup?.end; + const stepSetupEnd = trial.phases.agentExecution?.start; + if ( + stepSetupStart !== undefined && + stepSetupEnd !== undefined && + stepSetupEnd > stepSetupStart + ) { + const event = phaseEvent(rowId, "step_setup", { + start: stepSetupStart, + end: stepSetupEnd, + durationMs: (stepSetupEnd - stepSetupStart) * 1000, + }); + if (event) events.push(event); + } + + if (trial.browser?.start !== undefined && trial.browser.end !== undefined) { + const event = phaseEvent( + rowId, + "browser_session", + { + start: trial.browser.start, + end: trial.browser.end, + durationMs: (trial.browser.end - trial.browser.start) * 1000, + }, + { + timeoutSeconds: trial.browser.timeoutSeconds, + deletionVerified: trial.browser.deletionVerified, + }, + ); + if (event) events.push(event); + } + + const finalizeStart = trial.phases.verifier?.end; + const finalizeEnd = trial.metrics.end; + if ( + finalizeStart !== undefined && + finalizeEnd !== undefined && + finalizeEnd > finalizeStart + ) { + const event = phaseEvent(rowId, "finalize", { + start: finalizeStart, + end: finalizeEnd, + durationMs: (finalizeEnd - finalizeStart) * 1000, + }); + if (event) events.push(event); + } + return events; +} + function atifEvents(trial: BenchmarkTrial, rowId: string): BraintrustEvent[] { const events: BraintrustEvent[] = []; - const agentSteps = trajectorySteps(trial).filter( - (candidate) => candidate.source === "agent", - ); - for (const [stepIndex, step] of agentSteps.entries()) { + const steps = trajectorySteps(trial); + const sensitiveValues = [ + ...collectSensitiveValues(steps), + ...privateInfoValues(steps), + ]; + const agentExecutionId = uuidV5(`${rowId}:phase:agent_execution`); + let previousEnd = trial.phases.agentExecution?.start; + + for (const [trajectoryIndex, step] of steps.entries()) { + if (step.source !== "agent") continue; const stepId = step.step_id; - const stepKey = `${stepId ?? "missing"}:${stepIndex}`; + const stepKey = `${stepId ?? "missing"}:${trajectoryIndex}`; const llmId = uuidV5(`${rowId}:llm:${stepKey}`); - const start = step.timestamp - ? Date.parse(step.timestamp) / 1000 - : undefined; + const end = timestampSeconds(step.timestamp); + const start = + previousEnd === undefined || end === undefined + ? end + : Math.min(previousEnd, end); + if (end !== undefined) previousEnd = end; + const promptTokens = number(step.metrics?.prompt_tokens); + const completionTokens = number(step.metrics?.completion_tokens); const llmMetrics = Object.fromEntries( Object.entries({ start, - end: start, - prompt_tokens: number(step.metrics?.prompt_tokens), - completion_tokens: number(step.metrics?.completion_tokens), - cost_usd: number(step.metrics?.cost_usd), + end, + prompt_tokens: promptTokens, + prompt_cached_tokens: number(step.metrics?.cached_tokens), + completion_tokens: completionTokens, + tokens: + promptTokens === undefined || completionTokens === undefined + ? undefined + : promptTokens + completionTokens, + estimated_cost: number(step.metrics?.cost_usd), }).filter((entry): entry is [string, number] => entry[1] !== undefined), ); events.push({ id: llmId, span_id: llmId, root_span_id: rowId, - span_parents: [rowId], + span_parents: [ + trial.phases.agentExecution?.start !== undefined && + trial.phases.agentExecution.end !== undefined + ? agentExecutionId + : rowId, + ], span_attributes: { name: "agent", type: "llm" }, - created: step.timestamp, - output: redactValue(step.message), + created: timestampIso(start) ?? step.timestamp, + input: llmInput(steps, trajectoryIndex, sensitiveValues), + output: llmOutput(step, sensitiveValues), metadata: { phase: "agent_execution", + timing: "ATIF turn completion interval", stepId, - stepIndex, + trajectoryIndex, model: step.model_name ?? trial.model, }, metrics: llmMetrics, @@ -207,22 +434,34 @@ function atifEvents(trial: BenchmarkTrial, rowId: string): BraintrustEvent[] { const observation = step.observation?.results?.find( (result) => result.source_call_id === call.tool_call_id, ); + const toolName = call.function_name ?? "tool"; + const sessionId = + call.arguments !== null && typeof call.arguments === "object" + ? (call.arguments as Record).session_id + : undefined; + const sessionIdMatchesExpected = + typeof sessionId === "string" && trial.expectedBrowserSessionId + ? sessionId === trial.expectedBrowserSessionId + : undefined; events.push({ id: toolId, span_id: toolId, root_span_id: rowId, span_parents: [llmId], - span_attributes: { name: call.function_name ?? "tool", type: "tool" }, + span_attributes: { name: toolName, type: "tool" }, created: step.timestamp, - input: redactValue(call.arguments), - output: redactValue(observation?.content), + input: redactValueWithSecrets(call.arguments, sensitiveValues), + output: privateInfoRead(toolName, call.arguments) + ? REDACTED_PRIVATE_INFO + : redactValueWithSecrets(observation?.content, sensitiveValues), metadata: { phase: "agent_execution", + timing: "not available in ATIF", stepId, - stepIndex, + trajectoryIndex, toolCallId: call.tool_call_id, + sessionIdMatchesExpected, }, - metrics: start === undefined ? undefined : { start: start, end: start }, _is_merge: false, }); } @@ -249,7 +488,11 @@ export function buildExperimentEvents( span_parents: [], span_attributes: { name: trial.taskName, type: "eval" }, created: trial.startedAt, - input: { source: trial.source, taskName: trial.taskName }, + input: { + source: trial.source, + taskName: trial.taskName, + instruction: taskInstruction(trial), + }, output: { reward: primaryReward?.value, rewardKey: primaryReward?.key, @@ -262,9 +505,13 @@ export function buildExperimentEvents( metrics: metricRecord(trial), _is_merge: false, }); + events.push(...phaseEvents(trial, rowId)); events.push(...atifEvents(trial, rowId)); } } + for (const event of events) { + assertSafeToPublish(event, `event ${event.id}`); + } return events; } diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 851c7921..6547773d 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -1,7 +1,241 @@ +import ts from "typescript"; + const SECRET_NAME = /(API_KEY|TOKEN|SECRET|PASSWORD|PRIVATE_KEY|CREDENTIAL)/i; -const SENSITIVE_FIELD = - /(API_KEY|TOKEN|JWT|SECRET|PASSWORD|PRIVATE_KEY|CREDENTIAL|^COOKIE$|^SET-COOKIE$)/i; const REDACTED = "[REDACTED]"; +const REDACTED_PRIVATE_INFO = "[REDACTED_PRIVATE_INFO]"; + +const SENSITIVE_FIELDS = new Set([ + "api_key", + "access_token", + "auth_token", + "refresh_token", + "session_token", + "token", + "jwt", + "secret", + "password", + "private_key", + "credential", + "credentials", + "cookie", + "set_cookie", + "session_id", + "replay_id", + "cdp_url", + "cdp_ws_url", + "viewer_url", + "browser_live_view_url", + "email", + "phone", + "address", + "account_number", + "card_number", + "routing_number", + "ssn", + "tax_id", +]); + +function normalizedField(key: string): string { + return key + .trim() + .replace(/([a-z0-9])([A-Z])/g, "$1_$2") + .toLowerCase() + .replace(/-/g, "_"); +} + +function sensitiveField(key: string): boolean { + const normalized = normalizedField(key); + return ( + SENSITIVE_FIELDS.has(normalized) || + /(?:^|_)(?:api_key|access_token|auth_token|refresh_token|session_token|password|private_key|credential|secret(?:_key)?|session_id|replay_id|cdp_url|viewer_url)$/.test( + normalized, + ) + ); +} + +const SENSITIVE_QUERY_VALUE = /([?&])([a-z0-9_-]+)=([^&#\s]+)/gi; +const SENSITIVE_QUERY_ONLY_FIELDS = new Set(["auth", "code"]); + +interface SensitiveAssignment { + start: number; + end: number; + value: string; +} + +function structuredValueEnd(text: string, start: number): number { + const stack = [text[start]]; + let quote: string | undefined; + for (let cursor = start + 1; cursor < text.length; cursor += 1) { + const character = text[cursor]; + if (quote) { + if (character === "\\") cursor += 1; + else if (character === quote) quote = undefined; + continue; + } + if (character === '"' || character === "'") { + quote = character; + } else if (character === "{" || character === "[") { + stack.push(character); + } else if ( + (character === "}" && stack.at(-1) === "{") || + (character === "]" && stack.at(-1) === "[") + ) { + stack.pop(); + if (stack.length === 0) return cursor + 1; + } + } + return text.length; +} + +function skipHorizontalWhitespace(text: string, start: number): number { + let cursor = start; + while (text[cursor] === " " || text[cursor] === "\t") cursor += 1; + return cursor; +} + +function lineEndsAfter(text: string, start: number): boolean { + const cursor = skipHorizontalWhitespace(text, start); + return ( + cursor >= text.length || + text[cursor] === "\n" || + text[cursor] === "\r" || + (text[cursor] === "\\" && /[nr]/.test(text[cursor + 1] ?? "")) + ); +} + +function skipSnapshotAttributes(text: string, start: number): number { + let cursor = skipHorizontalWhitespace(text, start); + while (text[cursor] === "[") { + cursor = skipHorizontalWhitespace(text, structuredValueEnd(text, cursor)); + } + return cursor; +} + +function sensitiveAssignments(text: string): SensitiveAssignment[] { + const assignments: SensitiveAssignment[] = []; + for (let separator = 0; separator < text.length; separator += 1) { + if (text[separator] !== ":" && text[separator] !== "=") continue; + + let keyCursor = separator - 1; + while (/\s/.test(text[keyCursor] ?? "")) keyCursor -= 1; + const keyHasClosingQuote = + text[keyCursor] === '"' || text[keyCursor] === "'"; + if (keyHasClosingQuote) { + keyCursor -= 1; + while (text[keyCursor] === "\\") keyCursor -= 1; + } + const keyEnd = keyCursor + 1; + while (/[A-Za-z0-9_-]/.test(text[keyCursor] ?? "")) { + if (/[nrt]/.test(text[keyCursor] ?? "") && text[keyCursor - 1] === "\\") { + break; + } + keyCursor -= 1; + } + const keyStart = keyCursor + 1; + const key = text.slice(keyStart, keyEnd); + if (!key || !sensitiveField(key)) continue; + + let valueCursor = separator + 1; + while (/\s/.test(text[valueCursor] ?? "")) valueCursor += 1; + let slashCount = 0; + while (text[valueCursor + slashCount] === "\\") slashCount += 1; + let quote = text[valueCursor + slashCount]; + let start = valueCursor; + let end = valueCursor; + + if (quote === '"' || quote === "'") { + let openingSlashCount = 0; + for (let cursor = keyStart - 2; text[cursor] === "\\"; cursor -= 1) { + openingSlashCount += 1; + } + const quotedLabel = + !keyHasClosingQuote && + text[keyStart - 1] === quote && + openingSlashCount === slashCount; + if (quotedLabel && lineEndsAfter(text, valueCursor + slashCount + 1)) { + continue; + } + if (quotedLabel) { + valueCursor = skipSnapshotAttributes( + text, + valueCursor + slashCount + 1, + ); + if (lineEndsAfter(text, valueCursor)) continue; + if (text[valueCursor] === ":" || text[valueCursor] === "=") { + valueCursor = skipHorizontalWhitespace(text, valueCursor + 1); + } + slashCount = 0; + while (text[valueCursor + slashCount] === "\\") slashCount += 1; + quote = text[valueCursor + slashCount]; + start = valueCursor; + end = valueCursor; + } + } + + if (quote === '"' || quote === "'") { + start = valueCursor + slashCount + 1; + end = text.length; + for (let cursor = start; cursor < text.length; cursor += 1) { + if (text[cursor] !== quote) continue; + let precedingSlashes = 0; + for ( + let slashCursor = cursor - 1; + text[slashCursor] === "\\"; + slashCursor -= 1 + ) { + precedingSlashes += 1; + } + if (precedingSlashes === slashCount) { + end = cursor - slashCount; + separator = cursor; + break; + } + } + if (end === text.length) separator = text.length; + } else if (quote === "{" || quote === "[") { + start = valueCursor; + end = structuredValueEnd(text, start); + separator = end - 1; + } else { + while (end < text.length) { + const character = text[end] ?? ""; + if ( + /[\s,\}&;]/.test(character) || + (character === "\\" && /[nrt]/.test(text[end + 1] ?? "")) + ) { + break; + } + if ( + (character === '"' || character === "'") && + !( + character === "'" && + /[A-Za-z0-9]/.test(text[end - 1] ?? "") && + /[A-Za-z0-9]/.test(text[end + 1] ?? "") + ) + ) { + break; + } + end += 1; + } + separator = Math.max(separator, end - 1); + } + + assignments.push({ start, end, value: text.slice(start, end) }); + } + return assignments; +} + +function redactSensitiveAssignments(value: string): string { + let redacted = value; + for (const assignment of sensitiveAssignments(value).reverse()) { + redacted = `${redacted.slice(0, assignment.start)}${REDACTED}${redacted.slice(assignment.end)}`; + } + return redacted.replace(SENSITIVE_QUERY_VALUE, (match, prefix, key) => + sensitiveField(key) || SENSITIVE_QUERY_ONLY_FIELDS.has(normalizedField(key)) + ? `${prefix}${key}=${REDACTED}` + : match, + ); +} function secretValues(): string[] { return Object.entries(process.env) @@ -12,24 +246,103 @@ function secretValues(): string[] { .sort((left, right) => right.length - left.length); } -export function redactString(value: string, maxLength = 20_000): string { +const TYPING_METHODS = new Set([ + "fill", + "type", + "pressSequentially", + "insertText", +]); + +interface TypedLiteral { + start: number; + end: number; +} + +function typedLiterals(value: string): TypedLiteral[] { + const callStart = /^\.([A-Za-z]+)\s*\(/; + const literals: TypedLiteral[] = []; + let cursor = 0; + + while (cursor < value.length) { + const match = value.slice(cursor).match(callStart); + if (!match || !TYPING_METHODS.has(match[1])) { + cursor += 1; + continue; + } + + let callCursor = cursor + match[0].length; + let depth = 1; + let lastLiteral: TypedLiteral | undefined; + while (callCursor < value.length && depth > 0) { + const callQuote = value[callCursor]; + if (callQuote === '"' || callQuote === "'" || callQuote === "`") { + const start = callCursor; + callCursor += 1; + while (callCursor < value.length) { + if (value[callCursor] === "\\") { + callCursor += 2; + } else if (value[callCursor] === callQuote) { + callCursor += 1; + break; + } else { + callCursor += 1; + } + } + lastLiteral = { start, end: callCursor }; + continue; + } + if (value.startsWith("//", callCursor)) { + const newline = value.indexOf("\n", callCursor + 2); + callCursor = newline === -1 ? value.length : newline + 1; + continue; + } + if (value.startsWith("/*", callCursor)) { + const commentEnd = value.indexOf("*/", callCursor + 2); + callCursor = commentEnd === -1 ? value.length : commentEnd + 2; + continue; + } + if (value[callCursor] === "(") depth += 1; + if (value[callCursor] === ")") depth -= 1; + callCursor += 1; + } + + if (lastLiteral) literals.push(lastLiteral); + cursor = callCursor; + } + return literals; +} + +function typedCallValues(value: string): string[] { + return typedLiterals(value).map((literal) => + value.slice(literal.start + 1, literal.end - 1), + ); +} + +function redactTypedLiterals(value: string): string { let redacted = value; - for (const secret of secretValues()) { - redacted = redacted.split(secret).join(REDACTED); + for (const literal of typedLiterals(value).sort( + (left, right) => right.start - left.start, + )) { + const quote = value[literal.start]; + redacted = `${redacted.slice(0, literal.start)}${quote}${REDACTED}${quote}${redacted.slice(literal.end)}`; + } + return redacted; +} + +export function redactStringWithSecrets( + value: string, + additionalSecrets: string[], + maxLength = 20_000, +): string { + let redacted = value; + for (const secret of [...secretValues(), ...additionalSecrets]) { + if (secret.length >= 4) redacted = redacted.split(secret).join(REDACTED); } - redacted = redacted + redacted = redactSensitiveAssignments(redactTypedLiterals(redacted)) .replace(/\bBearer\s+[A-Za-z0-9._~+/=-]+/gi, `Bearer ${REDACTED}`) .replace(/\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\b/g, REDACTED) .replace(/\b(?:sk|pk|bt|kapi|whsec)[-_][A-Za-z0-9_-]{12,}\b/gi, REDACTED) - .replace( - /(["']?(?:api[_-]?key|access[_-]?token|credential|jwt|password|secret|token)["']?\s*[:=]\s*["']?)[^"'\s,}&]+/gi, - `$1${REDACTED}`, - ) - .replace( - /([?&](?:api[_-]?key|access[_-]?token|auth|code|credential|jwt|password|secret|session[_-]?token|token)=)[^&#\s]+/gi, - `$1${REDACTED}`, - ) .replace(/(\b(?:cookie|set-cookie)\s*:\s*)[^\r\n]+/gi, `$1${REDACTED}`) .replace(/(\/browser\/live\/)[^/?#\s]+/gi, `$1${REDACTED}`) .replace(/(wss?:\/\/)[^/@\s]+@/gi, `$1${REDACTED}@`) @@ -40,20 +353,299 @@ export function redactString(value: string, maxLength = 20_000): string { : redacted; } -export function redactValue(value: unknown, maxStringLength = 20_000): unknown { - if (typeof value === "string") return redactString(value, maxStringLength); +export function redactString(value: string, maxLength = 20_000): string { + return redactStringWithSecrets(value, [], maxLength); +} + +export function redactValueWithSecrets( + value: unknown, + additionalSecrets: string[], + maxStringLength = 20_000, +): unknown { + if (typeof value === "string") { + return redactStringWithSecrets(value, additionalSecrets, maxStringLength); + } if (Array.isArray(value)) { - return value.map((entry) => redactValue(entry, maxStringLength)); + return value.map((entry) => + redactValueWithSecrets(entry, additionalSecrets, maxStringLength), + ); } if (value !== null && typeof value === "object") { return Object.fromEntries( Object.entries(value as Record).map(([key, entry]) => [ key, - SENSITIVE_FIELD.test(key) + sensitiveField(key) ? REDACTED - : redactValue(entry, maxStringLength), + : redactValueWithSecrets(entry, additionalSecrets, maxStringLength), ]), ); } return value; } + +export function redactValue(value: unknown, maxStringLength = 20_000): unknown { + return redactValueWithSecrets(value, [], maxStringLength); +} + +export function collectSensitiveValues(value: unknown): string[] { + const values = new Set(); + const collectString = (text: string) => { + for (const assignment of sensitiveAssignments(text)) { + if (assignment.value.length >= 4) values.add(assignment.value); + if (/^[{[]/.test(assignment.value)) collectString(assignment.value); + } + for (const typedValue of typedCallValues(text)) { + if (typedValue.length >= 4) values.add(typedValue); + } + for (const match of text.matchAll( + /["'](?:id|name|type)["']\s*:\s*["'][^"']*password[^"']*["'][\s\S]{0,300}?["']value["']\s*:\s*["']([^"']{4,})["']/gi, + )) { + values.add(match[1]); + } + }; + const visit = (entry: unknown, key?: string) => { + if (typeof entry === "string") { + if (key && sensitiveField(key) && entry.length >= 4) values.add(entry); + collectString(entry); + return; + } + if (Array.isArray(entry)) { + for (const item of entry) visit(item); + return; + } + if (entry !== null && typeof entry === "object") { + for (const [childKey, child] of Object.entries( + entry as Record, + )) { + visit(child, childKey); + } + } + }; + visit(value); + return [...values].sort((left, right) => right.length - left.length); +} + +const PRIVATE_INFO_CONTEXTS = new Set([ + "financial", + "government_ids", + "insurance", +]); +const PRIVATE_INFO_IDENTIFIERS = new Set([ + "number", + "number_formatted", + "sin", + "transit_number", +]); + +export function collectPrivateInfoValues(value: unknown): string[] { + const values = new Set(collectSensitiveValues(value)); + const visit = (entry: unknown, inSensitiveContext = false): void => { + if (Array.isArray(entry)) { + for (const item of entry) visit(item, inSensitiveContext); + return; + } + if (entry === null || typeof entry !== "object") return; + for (const [key, child] of Object.entries( + entry as Record, + )) { + const normalized = normalizedField(key); + const childContext = + inSensitiveContext || PRIVATE_INFO_CONTEXTS.has(normalized); + if ( + childContext && + PRIVATE_INFO_IDENTIFIERS.has(normalized) && + typeof child === "string" && + child.length >= 4 + ) { + values.add(child); + } + visit(child, childContext); + } + }; + visit(value); + return [...values].sort((left, right) => right.length - left.length); +} + +export function privateInfoRead(toolName: string, input: unknown): boolean { + if (!/(?:^|__)(?:exec_command|bash|read)$/i.test(toolName)) return false; + const fields = + input !== null && typeof input === "object" + ? (input as Record) + : {}; + const text = [ + fields.cmd, + fields.command, + fields.file_path, + fields.path, + typeof input === "string" ? input : undefined, + ] + .filter((entry) => entry !== undefined) + .map(String) + .join("\n"); + const paths = [ + ...text.matchAll( + /(?:^|[\s"'`=])(?:\.\/|\/(?:workspace\/)?|workspace\/)?my-info\/([^\s"'`;)]*)/g, + ), + ].map((match) => match[1]); + return paths.some( + (path) => + path.length === 0 || !/^kernel_browser\.json(?:$|[?#])/.test(path), + ); +} + +function assertTypedCallsRedacted(value: string): void { + for (const match of value.matchAll( + /\.(fill|type|pressSequentially|insertText)\s*\(/g, + )) { + const source = ts.createSourceFile( + "typed-call.ts", + `receiver${value.slice(match.index)}`, + ts.ScriptTarget.Latest, + true, + ts.ScriptKind.TS, + ); + let target: ts.CallExpression | undefined; + const findTarget = (node: ts.Node): void => { + if ( + !target && + ts.isCallExpression(node) && + ts.isPropertyAccessExpression(node.expression) && + node.expression.name.text === match[1] && + node.expression.expression.getText(source) === "receiver" + ) { + target = node; + return; + } + ts.forEachChild(node, findTarget); + }; + findTarget(source); + if (!target) { + throw new Error("Braintrust payload contains an unvalidated typing call"); + } + + const literals: ts.Node[] = []; + const collectLiterals = (node: ts.Node): void => { + if ( + ts.isStringLiteral(node) || + ts.isNoSubstitutionTemplateLiteral(node) || + ts.isTemplateExpression(node) + ) { + literals.push(node); + return; + } + ts.forEachChild(node, collectLiterals); + }; + for (const argument of target.arguments) collectLiterals(argument); + const literal = literals.at(-1); + if (!literal) continue; + const literalValue = + ts.isStringLiteral(literal) || ts.isNoSubstitutionTemplateLiteral(literal) + ? literal.text + : literal.getText(source).slice(1, -1); + if (literalValue !== REDACTED) { + throw new Error("Braintrust payload still contains a typed form value"); + } + } +} + +function assertSensitiveAssignmentsRedacted(value: string): void { + const assignmentStart = + /(?:^|\\[nrt]|[^A-Za-z0-9_-])(?:\\*["'])?([A-Za-z0-9_-]+)(?:\\*["'])?\s*[:=]\s*/gi; + for (const match of value.matchAll(assignmentStart)) { + if (!sensitiveField(match[1])) continue; + const remainderStart = (match.index ?? 0) + match[0].length; + const remainder = value.slice(remainderStart); + const valueQuote = remainder.match(/^(\\*)(["'])/); + const separator = Math.max( + match[0].lastIndexOf(":"), + match[0].lastIndexOf("="), + ); + const quotedLabel = + valueQuote && + [...match[0].slice(0, separator)].filter( + (character) => character === valueQuote[2], + ).length % + 2 === + 1; + if ( + quotedLabel && + lineEndsAfter( + value, + (match.index ?? 0) + match[0].length + valueQuote[0].length, + ) + ) { + continue; + } + let unquoted = remainder.replace(/^(?:\\*["'])?/, ""); + if (quotedLabel && valueQuote) { + let valueStart = skipSnapshotAttributes( + value, + remainderStart + valueQuote[0].length, + ); + if (lineEndsAfter(value, valueStart)) continue; + if (value[valueStart] === ":" || value[valueStart] === "=") { + valueStart = skipHorizontalWhitespace(value, valueStart + 1); + } + unquoted = value.slice(valueStart).replace(/^(?:\\*["'])?/, ""); + } + if (!unquoted.startsWith(REDACTED)) { + throw new Error( + "Braintrust payload still contains a sensitive field value", + ); + } + const afterRedaction = unquoted + .slice(REDACTED.length) + .replace(/^\\*["']/, ""); + if (/^[A-Za-z0-9]/.test(afterRedaction)) { + throw new Error( + "Braintrust payload still contains data after a redacted field value", + ); + } + } +} + +function assertSafeString(value: string): void { + if (/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/i.test(value)) { + throw new Error("Braintrust payload still contains an email address"); + } + assertTypedCallsRedacted(value); + assertSensitiveAssignmentsRedacted(value); + for (const secret of secretValues()) { + if (value.includes(secret)) { + throw new Error("Braintrust payload still contains a configured secret"); + } + } +} + +export function assertSafeToPublish(value: unknown, path = "$"): void { + if (typeof value === "string") { + try { + assertSafeString(value); + } catch (error) { + const message = error instanceof Error ? error.message : String(error); + throw new Error(`${message} at ${path}`); + } + return; + } + if (Array.isArray(value)) { + for (const [index, entry] of value.entries()) { + assertSafeToPublish(entry, `${path}[${index}]`); + } + return; + } + if (value !== null && typeof value === "object") { + for (const [key, entry] of Object.entries( + value as Record, + )) { + const childPath = /^[A-Za-z_$][A-Za-z0-9_$]*$/.test(key) + ? `${path}.${key}` + : `${path}[${JSON.stringify(key)}]`; + if (sensitiveField(key) && entry !== REDACTED) { + throw new Error(`Braintrust payload did not redact ${childPath}`); + } + assertSafeToPublish(entry, childPath); + } + } +} + +export { REDACTED, REDACTED_PRIVATE_INFO }; diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index a0774091..1182ea84 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -11,7 +11,13 @@ import { join } from "node:path"; import { buildExperimentEvents, publishBenchmark } from "./publish-braintrust"; import { renderMarkdown } from "./report"; import { readBenchmarkArm, selectPrimaryReward, summarizeArm } from "./results"; -import { redactString, redactValue } from "./redact"; +import { + assertSafeToPublish, + collectSensitiveValues, + privateInfoRead, + redactString, + redactValue, +} from "./redact"; import { assertProjectScopedCredential } from "./verify-project-scope"; const temporaryDirectories: string[] = []; @@ -66,6 +72,14 @@ function fixture(): string { }, started_at: "2026-01-01T00:00:00Z", finished_at: "2026-01-01T00:01:00Z", + environment_setup: { + started_at: "2026-01-01T00:00:00Z", + finished_at: "2026-01-01T00:00:05Z", + }, + agent_setup: { + started_at: "2026-01-01T00:00:05Z", + finished_at: "2026-01-01T00:00:10Z", + }, step_results: [ { agent_result: { @@ -74,6 +88,14 @@ function fixture(): string { n_output_tokens: 20, cost_usd: 0.01, }, + agent_execution: { + started_at: "2026-01-01T00:00:20Z", + finished_at: "2026-01-01T00:00:40Z", + }, + verifier: { + started_at: "2026-01-01T00:00:45Z", + finished_at: "2026-01-01T00:00:55Z", + }, }, ], }); @@ -81,19 +103,40 @@ function fixture(): string { steps: [ { step_id: 1, + source: "system", + timestamp: "2026-01-01T00:00:20Z", + message: "system prompt", + }, + { + step_id: 2, + source: "user", + timestamp: "2026-01-01T00:00:20Z", + message: "perform the task", + }, + { + step_id: 3, source: "agent", - timestamp: "2026-01-01T00:00:01Z", - message: "working", + timestamp: "2026-01-01T00:00:21Z", + message: "", tool_calls: [ { tool_call_id: "call-1", function_name: "execute_playwright_code", - arguments: { code: "return 'done'" }, + arguments: { + session_id: "session-123", + code: "await page.locator('#password').fill('secret-password'); return 'done'", + }, }, ], observation: { results: [{ source_call_id: "call-1", content: "done" }], }, + metrics: { + prompt_tokens: 100, + cached_tokens: 80, + completion_tokens: 20, + cost_usd: 0.01, + }, }, ], }); @@ -101,6 +144,20 @@ function fixture(): string { kernel_mcp_server_sha: "server-sha", clawbench_source_sha: "clawbench-sha", }); + writeJson(join(success, "steps/run/verifier/kernel-mcp-result.json"), { + expected_session_id: "session-123", + }); + writeJson( + join(success, "steps/run/verifier/data/kernel-browser-lifecycle.json"), + { + timeout_seconds: 1920, + deletion_verified: true, + events: [ + { event: "browser_created", ts: 1767225620 }, + { event: "browser_deleted", ts: 1767225640 }, + ], + }, + ); const failed = join(root, "task-two__def"); writeJson(join(failed, "result.json"), { @@ -154,6 +211,30 @@ describe("Harbor result ingestion", () => { }); }); + test("classifies ungraded step setup failures as infrastructure", () => { + const root = fixture(); + const failedPath = join(root, "task-two__def", "result.json"); + const failed = JSON.parse(readFileSync(failedPath, "utf8")) as Record< + string, + unknown + >; + delete failed.exception_info; + failed.verifier_result = null; + failed.step_results = [ + { + exception_info: { + exception_type: "RuntimeError", + exception_message: "Step setup exited with code 1", + }, + }, + ]; + writeJson(failedPath, failed); + + const arm = readBenchmarkArm({ name: "candidate", path: root }); + expect(arm.trials[1].errorClass).toBe("infra"); + expect(arm.trials[1].error).toContain("Step setup exited with code 1"); + }); + test("summarizes against the intended task denominator", () => { const summary = summarizeArm( readBenchmarkArm({ name: "candidate", path: fixture() }), @@ -207,10 +288,14 @@ describe("Harbor result ingestion", () => { expect( first.filter((event) => event.span_attributes.type === "tool"), ).toHaveLength(1); + expect( + first.filter((event) => event.span_attributes.type === "task"), + ).toHaveLength(7); const root = first.find((event) => event.span_attributes.type === "eval"); expect(root?.input).toEqual({ source: "clawbench-v2", taskName: "v2-task-one", + instruction: "perform the task", }); expect(root?.span_parents).toEqual([]); const infra = first.find( @@ -229,6 +314,85 @@ describe("Harbor result ingestion", () => { reward: 1, rewardKey: "reward_lenient", }); + const llm = first.find((event) => event.span_attributes.type === "llm"); + expect(llm?.input).toEqual([ + { + source: "system", + message: "system prompt", + toolCalls: [], + observations: [], + }, + { + source: "user", + message: "perform the task", + toolCalls: [], + observations: [], + }, + ]); + expect(llm?.output).toMatchObject({ + message: "", + toolCalls: [ + { + name: "execute_playwright_code", + arguments: { + session_id: "[REDACTED]", + code: "await page.locator('#password').fill('[REDACTED]'); return 'done'", + }, + }, + ], + }); + const tool = first.find((event) => event.span_attributes.type === "tool"); + expect(tool?.metadata).toMatchObject({ + sessionIdMatchesExpected: true, + }); + const browser = first.find( + (event) => event.span_attributes.name === "browser_session", + ); + expect(browser?.metadata).toMatchObject({ + timeoutSeconds: 1920, + deletionVerified: true, + }); + expect(llm?.metrics).toMatchObject({ + start: Date.parse("2026-01-01T00:00:20Z") / 1000, + end: Date.parse("2026-01-01T00:00:21Z") / 1000, + prompt_tokens: 100, + prompt_cached_tokens: 80, + completion_tokens: 20, + tokens: 120, + estimated_cost: 0.01, + }); + expect(root?.metrics).toMatchObject({ + prompt_tokens: 100, + prompt_cached_tokens: 80, + completion_tokens: 20, + tokens: 120, + estimated_cost: 0.01, + }); + expect(root?.metrics).not.toHaveProperty("input_tokens"); + expect(root?.metrics).not.toHaveProperty("cost_usd"); + }); + + test("retains row cost when ATIF has no per-turn cost", () => { + const root = fixture(); + const trajectoryPath = join( + root, + "task-one__abc", + "steps/run/agent/trajectory.json", + ); + const trajectory = JSON.parse(readFileSync(trajectoryPath, "utf8")) as { + steps: Array<{ metrics?: Record }>; + }; + delete trajectory.steps[2].metrics?.cost_usd; + writeJson(trajectoryPath, trajectory); + + const events = buildExperimentEvents( + [readBenchmarkArm({ name: "candidate", path: root })], + "row-cost-fallback", + ); + const row = events.find((event) => event.span_attributes.type === "eval"); + const llm = events.find((event) => event.span_attributes.type === "llm"); + expect(row?.metrics).toMatchObject({ estimated_cost: 0.01 }); + expect(llm?.metrics).not.toHaveProperty("estimated_cost"); }); test("re-publishes the same rows and spans by deterministic ID", async () => { @@ -402,6 +566,121 @@ describe("Harbor result ingestion", () => { ).filter((event) => event.span_attributes.type === "llm"); expect(new Set(events.map((event) => event.id)).size).toBe(2); }); + + test("includes context added after the previous agent turn", () => { + const root = fixture(); + writeJson(join(root, "task-one__abc", "steps/run/agent/trajectory.json"), { + steps: [ + { source: "system", message: "system prompt" }, + { source: "user", message: "first request" }, + { source: "agent", message: "first response" }, + { source: "user", message: "follow-up" }, + { source: "system", message: "updated constraint" }, + { source: "agent", message: "second response" }, + ], + }); + const llmEvents = buildExperimentEvents( + [readBenchmarkArm({ name: "candidate", path: root })], + "multi-turn-context", + ).filter((event) => event.span_attributes.type === "llm"); + + expect(llmEvents[1].input).toEqual([ + { + source: "agent", + message: "first response", + toolCalls: [], + observations: [], + }, + { + source: "user", + message: "follow-up", + toolCalls: [], + observations: [], + }, + { + source: "system", + message: "updated constraint", + toolCalls: [], + observations: [], + }, + ]); + }); + + test("keeps login-page snapshots publishable", () => { + const root = fixture(); + const trajectoryPath = join( + root, + "task-one__abc", + "steps/run/agent/trajectory.json", + ); + const trajectory = JSON.parse(readFileSync(trajectoryPath, "utf8")) as { + steps: Array<{ + observation?: { results?: Array<{ content?: string }> }; + }>; + }; + trajectory.steps[2].observation!.results![0].content = JSON.stringify({ + success: true, + result: `- text: "Email:" +- textbox "Email:" +- text: "Password:" +- textbox "Password:" [disabled] [ref=e12]: Filled2Pass +Password: don't reuse one from another site`, + }); + writeJson(trajectoryPath, trajectory); + + const events = buildExperimentEvents( + [readBenchmarkArm({ name: "candidate", path: root })], + "login-snapshot", + ); + const published = JSON.stringify(events); + expect(published).toContain("Password:"); + expect(published).toContain("[REDACTED] reuse one from another site"); + expect(published).not.toContain("Filled2Pass"); + expect(published).not.toContain("don't"); + expect(() => assertSafeToPublish(events)).not.toThrow(); + }); + + test("harvests only sensitive values from private-info output", () => { + const root = fixture(); + writeJson(join(root, "task-one__abc", "steps/run/agent/trajectory.json"), { + steps: [ + { source: "user", message: "perform the task" }, + { + source: "agent", + message: "", + tool_calls: [ + { + tool_call_id: "private-read", + function_name: "Read", + arguments: { file_path: "/my-info/personal.json" }, + }, + ], + observation: { + results: [ + { + source_call_id: "private-read", + content: + '{"name":"Close Window","width":"1208","account_number":"12345678","government_ids":{"passport":{"number":"JK456789"},"drivers_license":{"number":"G4567-89018-05501"},"health_card":{"number":"6789-012-345"},"sin":"472-345-678"},"financial":{"bank_accounts":[{"transit_number":"10202"}],"credit_cards":[{"number":"4519873424604532"}]}}', + }, + ], + }, + }, + { + source: "agent", + message: + "Close Window width 1208 account 12345678 passport JK456789 licence G4567-89018-05501 health 6789-012-345 sin 472-345-678 transit 10202 card 4519873424604532", + }, + ], + }); + const llmEvents = buildExperimentEvents( + [readBenchmarkArm({ name: "candidate", path: root })], + "private-info-values", + ).filter((event) => event.span_attributes.type === "llm"); + + expect(llmEvents[1].output).toBe( + "Close Window width 1208 account [REDACTED] passport [REDACTED] licence [REDACTED] health [REDACTED] sin [REDACTED] transit [REDACTED] card [REDACTED]", + ); + }); }); describe("Braintrust redaction", () => { @@ -409,24 +688,223 @@ describe("Braintrust redaction", () => { process.env.TEST_API_KEY = "super-secret-value"; expect( redactString( - 'Bearer super-secret-value sk-proj-abcdefghijklmnop?access_token=visible&token=plain&jwt=opaque "password":"generated-password" Cookie: session=visible\nhttps://example.com/browser/live/replay-slug user@example.com', + 'Bearer super-secret-value sk-proj-abcdefghijklmnop?access_token=visible&token=plain&jwt=opaque "password":"generated-password" "session_id":"session-123" Cookie: session=visible\nhttps://example.com/browser/live/replay-slug user@example.com await page.locator("#password").fill("typed-password"); await page.fill("#password", "two-arg-secret")', ), ).toBe( - 'Bearer [REDACTED] [REDACTED]?access_token=[REDACTED]&token=[REDACTED]&jwt=[REDACTED] "password":"[REDACTED]" Cookie: [REDACTED]\nhttps://example.com/browser/live/[REDACTED] [REDACTED_EMAIL]', + 'Bearer [REDACTED] [REDACTED]?access_token=[REDACTED]&token=[REDACTED]&jwt=[REDACTED] "password":"[REDACTED]" "session_id":"[REDACTED]" Cookie: [REDACTED]\nhttps://example.com/browser/live/[REDACTED] [REDACTED_EMAIL] await page.locator("#password").fill("[REDACTED]"); await page.fill("#password", "[REDACTED]")', ); - expect( - redactValue({ - api_key: "visible", - Cookie: "session=visible", - nested: ["bt-abcdefghijklmnop"], - }), - ).toEqual({ + const redacted = redactValue({ + api_key: "visible", + Cookie: "session=visible", + session_id: "session-123", + max_output_tokens: 1000, + nested: ["bt-abcdefghijklmnop"], + }); + expect(redacted).toEqual({ api_key: "[REDACTED]", Cookie: "[REDACTED]", + session_id: "[REDACTED]", + max_output_tokens: 1000, nested: ["[REDACTED]"], }); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + expect(() => + assertSafeToPublish({ code: "page.fill('still-visible')" }), + ).toThrow("typed form value"); + expect(() => + assertSafeToPublish({ + code: "page.fill('#password', 'still-visible')", + }), + ).toThrow("typed form value"); + expect(() => + assertSafeToPublish({ + nested: { code: "page.fill('#password', 'still-visible')" }, + }), + ).toThrow("$.nested.code"); + expect( + privateInfoRead("exec_command", { + cmd: "cat /my-info/email_credentials.json", + }), + ).toBe(true); + expect( + privateInfoRead("Bash", { + command: "cat ./my-info/alex_green_personal_info.json", + }), + ).toBe(true); + expect( + privateInfoRead("Read", { + file_path: "/workspace/my-info/email_credentials.json", + }), + ).toBe(true); + expect( + privateInfoRead("exec_command", { + cmd: "cat /my-info/kernel_browser.json", + }), + ).toBe(false); delete process.env.TEST_API_KEY; }); + + test("redacts normalized cookie headers and compound secret keys", () => { + const redacted = redactValue({ + "Set-Cookie": "session=visible", + client_secret: "client-value", + secret_key: "key-value", + webhook_secret: "webhook-value", + apiKey: "camel-api-value", + sessionId: "camel-session-value", + clientSecret: "camel-secret-value", + accountNumber: "camel-account-value", + }); + expect(redacted).toEqual({ + "Set-Cookie": "[REDACTED]", + client_secret: "[REDACTED]", + secret_key: "[REDACTED]", + webhook_secret: "[REDACTED]", + apiKey: "[REDACTED]", + sessionId: "[REDACTED]", + clientSecret: "[REDACTED]", + accountNumber: "[REDACTED]", + }); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + expect(() => + assertSafeToPublish({ "Set-Cookie": "session=visible" }), + ).toThrow("Set-Cookie"); + expect(() => + assertSafeToPublish({ client_secret: "client-value" }), + ).toThrow("client_secret"); + expect(() => + assertSafeToPublish({ clientSecret: "camel-secret-value" }), + ).toThrow("clientSecret"); + + const text = redactString( + `'client_secret': 'client-value'&webhook_secret=webhook-value; 'address': '123 Main St'; "apiKey":"camel-api-value"; "sessionId":"camel-session-value"; "clientSecret":"camel-secret-value"`, + ); + for (const secret of [ + "client-value", + "webhook-value", + "123 Main St", + "camel-api-value", + "camel-session-value", + "camel-secret-value", + ]) { + expect(text).not.toContain(secret); + } + }); + + test("redacts nested and unterminated sensitive assignments", () => { + const nested = + '{"success": true, "result": "Account created.\\nTemporary password: Hunter2Pass"}'; + const unterminated = + '{"text": "Step 1...' + "\\n".repeat(30) + "password: Unterminated2Pass"; + const escapedNested = + '{"success":true,"result":"{\\"password\\":\\"Escaped2Pass\\"}"}'; + const escapedValue = String.raw`password: \"EscapedValue2Pass\"`; + const escapedWrapped = String.raw`result: \"wrapper password: \\\"EscapedWrapped2Pass\\\"\"`; + + for (const [source, secret] of [ + [nested, "Hunter2Pass"], + [unterminated, "Unterminated2Pass"], + [escapedNested, "Escaped2Pass"], + [escapedValue, "EscapedValue2Pass"], + [escapedWrapped, "EscapedWrapped2Pass"], + ]) { + const redacted = redactString(source); + expect(redacted).not.toContain(secret); + expect(redacted).toContain("[REDACTED]"); + expect(collectSensitiveValues(source)).toContain(secret); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + expect(() => assertSafeToPublish(source)).toThrow( + "sensitive field value", + ); + } + expect(() => + assertSafeToPublish('password: [REDACTED]"StillVisible"'), + ).toThrow("data after a redacted field value"); + + const objectValue = + 'credentials: {"username":"alex","password":"Nested2Pass"}'; + const arrayValue = 'credentials: ["first", {"token":"NestedToken"}]'; + for (const source of ['password: ""', "token=", objectValue, arrayValue]) { + const redacted = redactString(source); + expect(redacted).toContain("[REDACTED]"); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + expect(() => assertSafeToPublish(source)).toThrow( + "sensitive field value", + ); + } + expect(collectSensitiveValues(objectValue)).toContain("Nested2Pass"); + expect(collectSensitiveValues(arrayValue)).toContain("NestedToken"); + }); + + test("keeps aria labels publishable and redacts prose values", () => { + const ariaSnapshot = `- text: "Email:" +- textbox "Email:" +- text: "Password:" +- textbox "Password:" [disabled] [ref=e12]`; + const escapedLabel = String.raw`- text: \"Password:\"`; + for (const label of [ariaSnapshot, escapedLabel]) { + expect(redactString(label)).toBe(label); + expect(collectSensitiveValues(label)).toEqual([]); + expect(() => assertSafeToPublish(label)).not.toThrow(); + } + + const prose = "Password: don't reuse one from another site"; + const redacted = redactString(prose); + expect(redacted).toBe("Password: [REDACTED] reuse one from another site"); + expect(collectSensitiveValues(prose)).toContain("don't"); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + + const filledSnapshot = + '- textbox "Password:" [disabled] [ref=e12]: Filled2Pass'; + const redactedSnapshot = redactString(filledSnapshot); + expect(redactedSnapshot).not.toContain("Filled2Pass"); + expect(collectSensitiveValues(filledSnapshot)).toContain("Filled2Pass"); + expect(() => assertSafeToPublish(redactedSnapshot)).not.toThrow(); + expect(() => assertSafeToPublish(filledSnapshot)).toThrow( + "sensitive field value", + ); + }); + + test("redacts complex Playwright typing calls and rejects originals", () => { + const calls = [ + `page.fill('#password', 'Str0ng)Pass!')`, + `page.type('#password', 'type)value')`, + `page.locator('#pw').fill('abc)def')`, + `page.locator('#pw').fill('forced)value', { force: true })`, + `page.locator('#pw').pressSequentially(\`multi\nline)pass\`)`, + `page.keyboard.insertText("typed)secret")`, + `page.fill(buildSelector('nested)selector'), 'last)value')`, + `// Fill in the user's password\nawait page.locator('#password').fill('Comment2Pass')`, + `/* we'll sign up now */\nawait page.fill('#password', 'Block2Pass')`, + `node -e 'await page.fill("#pw", "Shell2Pass")'`, + ]; + const source = calls.join(";\n"); + const redacted = redactString(source); + + for (const secret of [ + "Str0ng)Pass!", + "type)value", + "abc)def", + "forced)value", + "multi\nline)pass", + "typed)secret", + "last)value", + "Comment2Pass", + "Block2Pass", + "Shell2Pass", + ]) { + expect(redacted).not.toContain(secret); + expect(collectSensitiveValues({ code: source })).toContain(secret); + } + expect(redacted).toContain("buildSelector('nested)selector')"); + expect(redacted.match(/\[REDACTED\]/g)).toHaveLength(calls.length); + expect(() => assertSafeToPublish({ code: redacted })).not.toThrow(); + for (const call of calls) { + expect(() => assertSafeToPublish({ code: call })).toThrow( + "typed form value", + ); + } + }); }); describe("benchmark workflow hardening", () => { @@ -473,6 +951,7 @@ describe("benchmark workflow hardening", () => { "harbor_hypeman_version=${HARBOR_HYPEMAN_VERSION:-0.1.2}", ); expect(runner).toContain('--max-retries "${HARBOR_MAX_RETRIES:-5}"'); + expect(runner).toContain('bun "$benchmark_dir/verify-purelymail.ts"'); for (const exception of [ "APITimeoutError", "APIConnectionError", @@ -480,6 +959,8 @@ describe("benchmark workflow hardening", () => { "InternalServerError", "ConnectionRefusedError", "ExecProtocolError", + "AgentSetupTimeoutError", + "RuntimeError", ]) { expect(runner).toContain(`--retry-include ${exception}`); } diff --git a/benchmarks/harbor/results.ts b/benchmarks/harbor/results.ts index 41def571..c2024a05 100644 --- a/benchmarks/harbor/results.ts +++ b/benchmarks/harbor/results.ts @@ -20,6 +20,21 @@ export interface BenchmarkMetrics { end?: number; } +export interface BenchmarkPhase { + startedAt?: string; + finishedAt?: string; + start?: number; + end?: number; + durationMs?: number; +} + +export interface BenchmarkBrowserLifecycle { + timeoutSeconds?: number; + start?: number; + end?: number; + deletionVerified?: boolean; +} + export interface BenchmarkTrial { arm: string; id: string; @@ -35,6 +50,14 @@ export interface BenchmarkTrial { error?: string; errorClass?: "infra"; metrics: BenchmarkMetrics; + phases: { + environmentSetup?: BenchmarkPhase; + agentSetup?: BenchmarkPhase; + agentExecution?: BenchmarkPhase; + verifier?: BenchmarkPhase; + }; + browser?: BenchmarkBrowserLifecycle; + expectedBrowserSessionId?: string; kernelMcpSha?: string; clawbenchSha?: string; trajectoryPath?: string; @@ -154,6 +177,20 @@ function durationMs( return Number.isFinite(duration) && duration >= 0 ? duration : undefined; } +function phase(value: unknown): BenchmarkPhase | undefined { + const timing = object(value); + const startedAt = string(timing.started_at); + const finishedAt = string(timing.finished_at); + if (!startedAt && !finishedAt) return undefined; + return { + startedAt, + finishedAt, + start: isoSeconds(startedAt), + end: isoSeconds(finishedAt), + durationMs: durationMs(startedAt, finishedAt), + }; +} + function sumStepMetric( stepResults: unknown[], key: string, @@ -224,21 +261,40 @@ function parseTrial(arm: string, trialDir: string): BenchmarkTrial { const agentInfo = object(result.agent_info); const modelInfo = object(agentInfo.model_info); const steps = array(result.step_results); - const exception = result.exception_info; + const rewards = trialRewards(result, trialDir); const exceptionFile = join(trialDir, "exception.txt"); + const stepException = + Object.keys(rewards).length === 0 + ? steps.map((step) => object(step).exception_info).find(Boolean) + : undefined; const error = errorText( - exception ?? + result.exception_info ?? + stepException ?? (existsSync(exceptionFile) ? readFileSync(exceptionFile, "utf8") : undefined), ); - const rewards = trialRewards(result, trialDir); const startedAt = string(result.started_at); const finishedAt = string(result.finished_at); const trajectoryPath = join(trialDir, "steps/run/agent/trajectory.json"); const runManifest = readJsonIfPresent( join(trialDir, "steps/run/verifier/kernel-mcp/run-manifest.json"), ); + const kernelMcpResult = readJsonIfPresent( + join(trialDir, "steps/run/verifier/kernel-mcp-result.json"), + ); + const browserLifecycle = readJsonIfPresent( + join(trialDir, "steps/run/verifier/data/kernel-browser-lifecycle.json"), + ); + const browserEvents = array(browserLifecycle.events).map(object); + const browserCreated = browserEvents.find( + (event) => event.event === "browser_created", + ); + const browserDeleted = browserEvents.find( + (event) => event.event === "browser_deleted", + ); + const browserStart = number(browserCreated?.ts); + const browserEnd = number(browserDeleted?.ts); return { arm, @@ -274,6 +330,25 @@ function parseTrial(arm: string, trialDir: string): BenchmarkTrial { end: isoSeconds(finishedAt), ...trajectoryMetrics(trajectoryPath), }, + phases: { + environmentSetup: phase(result.environment_setup), + agentSetup: phase(result.agent_setup), + agentExecution: phase(object(steps.at(-1)).agent_execution), + verifier: phase(object(steps.at(-1)).verifier), + }, + browser: + browserStart === undefined && browserEnd === undefined + ? undefined + : { + timeoutSeconds: number(browserLifecycle.timeout_seconds), + start: browserStart, + end: browserEnd, + deletionVerified: + typeof browserLifecycle.deletion_verified === "boolean" + ? browserLifecycle.deletion_verified + : undefined, + }, + expectedBrowserSessionId: string(kernelMcpResult.expected_session_id), kernelMcpSha: string(runManifest.kernel_mcp_server_sha), clawbenchSha: string(runManifest.clawbench_source_sha), trajectoryPath: existsSync(trajectoryPath) ? trajectoryPath : undefined, diff --git a/benchmarks/harbor/verify-purelymail.test.ts b/benchmarks/harbor/verify-purelymail.test.ts new file mode 100644 index 00000000..d49bcd62 --- /dev/null +++ b/benchmarks/harbor/verify-purelymail.test.ts @@ -0,0 +1,46 @@ +import { describe, expect, test } from "bun:test"; +import { verifyPurelyMail } from "./verify-purelymail"; + +describe("PurelyMail benchmark preflight", () => { + test("creates and deletes a disposable account", async () => { + const requests: Array<{ endpoint: string; body: Record }> = + []; + const fetcher = (async (request, init) => { + requests.push({ + endpoint: new URL(String(request)).pathname.split("/").at(-1) ?? "", + body: JSON.parse(String(init?.body)), + }); + return Response.json({ type: "success" }); + }) as typeof fetch; + + await verifyPurelyMail("api-key", "example.test", fetcher); + + expect(requests.map((request) => request.endpoint)).toEqual([ + "createUser", + "deleteUser", + ]); + expect(requests[0].body).toMatchObject({ + domainName: "example.test", + enablePasswordReset: false, + sendWelcomeEmail: false, + }); + expect(requests[1].body.userName).toBe( + `${requests[0].body.userName}@example.test`, + ); + }); + + test("fails closed on API errors without printing credentials", async () => { + const fetcher = (async () => + Response.json({ + type: "error", + code: "invalidToken", + message: "Token not valid.", + })) as unknown as typeof fetch; + + await expect( + verifyPurelyMail("secret-api-key", "example.test", fetcher), + ).rejects.toThrow( + "PurelyMail createUser failed (invalidToken): Token not valid.", + ); + }); +}); diff --git a/benchmarks/harbor/verify-purelymail.ts b/benchmarks/harbor/verify-purelymail.ts new file mode 100644 index 00000000..8c8204fa --- /dev/null +++ b/benchmarks/harbor/verify-purelymail.ts @@ -0,0 +1,80 @@ +#!/usr/bin/env bun +import { randomBytes, randomUUID } from "node:crypto"; + +interface PurelyMailResponse { + type?: string; + code?: string; + message?: string; +} + +function env(name: string): string { + const value = process.env[name]?.trim(); + if (!value) throw new Error(`${name} is required`); + return value; +} + +async function request( + fetcher: typeof fetch, + apiKey: string, + endpoint: string, + body: Record, +): Promise { + const response = await fetcher(`https://purelymail.com/api/v0/${endpoint}`, { + method: "POST", + headers: { + "Purelymail-Api-Token": apiKey, + "Content-Type": "application/json", + }, + body: JSON.stringify(body), + }); + if (!response.ok) { + throw new Error(`PurelyMail ${endpoint} returned HTTP ${response.status}`); + } + const result = (await response.json()) as PurelyMailResponse; + if (!result || typeof result !== "object") { + throw new Error(`PurelyMail ${endpoint} returned an invalid response`); + } + if (result.type === "error") { + const code = result.code ? ` (${result.code})` : ""; + const message = result.message ? `: ${result.message}` : ""; + throw new Error(`PurelyMail ${endpoint} failed${code}${message}`); + } + return result; +} + +export async function verifyPurelyMail( + apiKey: string, + domain: string, + fetcher: typeof fetch = fetch, +): Promise { + const local = `cbpreflight${randomUUID().replace(/-/g, "").slice(0, 12)}`; + const email = `${local}@${domain}`; + const password = randomBytes(18).toString("base64url"); + let created = false; + try { + await request(fetcher, apiKey, "createUser", { + userName: local, + domainName: domain, + password, + enablePasswordReset: false, + sendWelcomeEmail: false, + }); + created = true; + } finally { + if (created) { + await request(fetcher, apiKey, "deleteUser", { userName: email }); + } + } +} + +async function main(): Promise { + await verifyPurelyMail(env("PURELY_MAIL_API_KEY"), env("PURELY_MAIL_DOMAIN")); + process.stdout.write("PurelyMail preflight passed\n"); +} + +if (import.meta.main) { + main().catch((error) => { + console.error(error instanceof Error ? error.message : error); + process.exitCode = 1; + }); +}