From 003080aaeecaf798df78f2e8f40658f7f25c4802 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 2 Sep 2026 01:46:06 +0000 Subject: [PATCH 01/14] Fix benchmark trace publication --- benchmarks/harbor/README.md | 8 +- benchmarks/harbor/clawbench/run.sh | 4 + benchmarks/harbor/publish-braintrust.ts | 287 ++++++++++++++++++-- benchmarks/harbor/redact.ts | 195 ++++++++++++- benchmarks/harbor/results.test.ts | 179 +++++++++++- benchmarks/harbor/results.ts | 81 +++++- benchmarks/harbor/verify-purelymail.test.ts | 46 ++++ benchmarks/harbor/verify-purelymail.ts | 80 ++++++ 8 files changed, 822 insertions(+), 58 deletions(-) create mode 100644 benchmarks/harbor/verify-purelymail.test.ts create mode 100644 benchmarks/harbor/verify-purelymail.ts diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index 7b7b2ce2..0e3d0bb5 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -57,7 +57,9 @@ Codex defaults to version `0.120.0` with `gpt-5.6-luna`. Claude Code defaults to Single-task runs have a 40-minute wall-clock limit. Full-suite runs default to six hours. Set `HARBOR_BENCHMARK_TIMEOUT` to override either limit. Set `HARBOR_JOBS_DIR` to choose where Harbor writes results. -The runner retries a whole isolated trial up to five times for transient Hypeman connection, timeout, and exec-stream failures. Set `HARBOR_MAX_RETRIES` to override that limit. Per-request SDK retries remain disabled because transparently retrying instance or image creation can duplicate a request whose first response was lost. +Before creating the Harbor dataset, the runner creates and deletes one disposable PurelyMail account. API errors stop the run before trials begin, so missing email accounts cannot silently become task failures. + +The runner retries a whole isolated trial up to five times for transient Hypeman connection, timeout, exec-stream, and agent/task setup failures. Set `HARBOR_MAX_RETRIES` to override that limit. Per-request SDK retries remain disabled because transparently retrying instance or image creation can duplicate a request whose first response was lost. ## GitHub Actions @@ -102,6 +104,8 @@ BRAINTRUST_PROJECT=kernel-mcp-server-benchmarks \ --arm baseline=/path/to/baseline-job ``` -The experiment name and deterministic row/span IDs make it safe to publish the same job directories again. Re-publication replaces the rows and refreshes experiment metadata. Rows contain task identity, numeric rewards, provenance, bounded errors, timing, token, call, and cost metrics. ATIF agent/tool activity is attached as child spans after secret redaction. Task instructions, ground truth, browser session URLs, and recordings are not placed on experiment rows or public pull-request comments. +The experiment name and deterministic row/span IDs make it safe to publish the same job directories again. Re-publication replaces the rows and refreshes experiment metadata. Rows contain task identity, the redacted instruction, numeric rewards, provenance, bounded errors, and trial timing. Setup, browser lifetime, execution, verification, and finalization are separate timeline spans; the browser span records its timeout and deletion status without its identifiers. ATIF turns include their preceding context, structured tool calls, cached-token metrics, and inferred turn intervals. Browser tool spans record whether the supplied session ID matched the trial's expected session without publishing either ID. Tool durations remain unspecified because ATIF does not record them. + +The publisher redacts typed form values, configured secrets, credentials, email addresses, browser session/replay IDs, private-info file contents, and provider URLs. It validates the final payload and aborts before creating an experiment if sensitive content remains. Ground truth, recordings, and raw Harbor jobs are never published. Reports suppress comparison deltas when either arm has an infrastructure failure or ungraded trial. The workflow fails unless every intended trial is graded, while still retaining the incomplete report for diagnosis. diff --git a/benchmarks/harbor/clawbench/run.sh b/benchmarks/harbor/clawbench/run.sh index 1feda7e2..04e84175 100755 --- a/benchmarks/harbor/clawbench/run.sh +++ b/benchmarks/harbor/clawbench/run.sh @@ -48,6 +48,8 @@ set +a : "${PURELY_MAIL_API_KEY:?PURELY_MAIL_API_KEY is required}" : "${PURELY_MAIL_DOMAIN:?PURELY_MAIL_DOMAIN is required}" +bun "$benchmark_dir/verify-purelymail.ts" + case "$agent" in claude-code) if [[ -z "${ANTHROPIC_API_KEY:-}" && -z "${ANTHROPIC_AUTH_TOKEN:-}" && -z "${CLAUDE_CODE_OAUTH_TOKEN:-}" ]]; then @@ -169,5 +171,7 @@ timeout --signal=INT --kill-after=30s "${HARBOR_BENCHMARK_TIMEOUT:-$default_time --retry-include InternalServerError \ --retry-include ConnectionRefusedError \ --retry-include ExecProtocolError \ + --retry-include AgentSetupTimeoutError \ + --retry-include RuntimeError \ --delete \ --yes diff --git a/benchmarks/harbor/publish-braintrust.ts b/benchmarks/harbor/publish-braintrust.ts index c6fa1c1b..4a8ad0d6 100644 --- a/benchmarks/harbor/publish-braintrust.ts +++ b/benchmarks/harbor/publish-braintrust.ts @@ -4,13 +4,22 @@ import { mkdirSync, readFileSync, writeFileSync } from "node:fs"; import { dirname } from "node:path"; import { type BenchmarkArm, + type BenchmarkPhase, type BenchmarkTrial, parseArmSpec, readBenchmarkArm, selectPrimaryReward, summarizeArm, } from "./results"; -import { redactString, redactValue } from "./redact"; +import { + assertSafeToPublish, + collectSensitiveValues, + privateInfoRead, + redactString, + redactValue, + redactValueWithSecrets, + REDACTED_PRIVATE_INFO, +} from "./redact"; interface CliOptions { arms: string[]; @@ -41,7 +50,7 @@ interface BraintrustEvent { span_id: string; root_span_id: string; span_parents: string[]; - span_attributes: { name: string; type: "eval" | "llm" | "tool" }; + span_attributes: { name: string; type: "eval" | "llm" | "tool" | "task" }; created?: string; input?: unknown; output?: unknown; @@ -127,16 +136,19 @@ function metricRecord(trial: BenchmarkTrial): Record { Object.entries({ start: trial.metrics.start, end: trial.metrics.end, - input_tokens: trial.metrics.inputTokens, - cached_tokens: trial.metrics.cacheTokens, - output_tokens: trial.metrics.outputTokens, - cost_usd: trial.metrics.costUsd, duration_ms: trial.metrics.durationMs, tool_calls: trial.metrics.toolCalls, }).filter((entry): entry is [string, number] => entry[1] !== undefined), ); } +function taskInstruction(trial: BenchmarkTrial): unknown { + const userSteps = trajectorySteps(trial).filter( + (step) => step.source === "user" && step.message !== undefined, + ); + return redactValue(userSteps.at(-1)?.message); +} + function trialMetadata(trial: BenchmarkTrial): Record { return { trialName: trial.trialName, @@ -161,24 +173,220 @@ function trialMetadata(trial: BenchmarkTrial): Record { }; } +function timestampSeconds(value?: string): number | undefined { + if (!value) return undefined; + const millis = Date.parse(value); + return Number.isFinite(millis) ? millis / 1000 : undefined; +} + +function timestampIso(value?: number): string | undefined { + return value === undefined ? undefined : new Date(value * 1000).toISOString(); +} + +function stepContext( + step: AtifStep, + sensitiveValues: string[], +): Record { + const calls = step.tool_calls ?? []; + return { + source: step.source, + message: redactValueWithSecrets(step.message, sensitiveValues), + toolCalls: calls.map((call) => ({ + name: call.function_name ?? "tool", + arguments: redactValueWithSecrets(call.arguments, sensitiveValues), + })), + observations: (step.observation?.results ?? []).map((result) => { + const call = calls.find( + (candidate) => candidate.tool_call_id === result.source_call_id, + ); + const toolName = call?.function_name ?? "tool"; + return { + source_call_id: result.source_call_id, + content: + call && privateInfoRead(toolName, call.arguments) + ? REDACTED_PRIVATE_INFO + : redactValueWithSecrets(result.content, sensitiveValues), + }; + }), + }; +} + +function llmInput( + steps: AtifStep[], + stepIndex: number, + sensitiveValues: string[], +): unknown { + const prior = steps.slice(0, stepIndex); + let previousAgent = -1; + for (let index = prior.length - 1; index >= 0; index -= 1) { + if (prior[index].source === "agent") { + previousAgent = index; + break; + } + } + const context = + previousAgent === -1 + ? prior + : prior.slice(previousAgent, previousAgent + 1); + return context.map((step) => stepContext(step, sensitiveValues)); +} + +function llmOutput(step: AtifStep, sensitiveValues: string[]): unknown { + if ((step.tool_calls ?? []).length === 0) { + return redactValueWithSecrets(step.message, sensitiveValues); + } + return { + message: redactValueWithSecrets(step.message, sensitiveValues), + toolCalls: (step.tool_calls ?? []).map((call) => ({ + name: call.function_name ?? "tool", + arguments: redactValueWithSecrets(call.arguments, sensitiveValues), + })), + }; +} + +function privateInfoValues(steps: AtifStep[]): string[] { + const values = new Set(); + for (const step of steps) { + for (const call of step.tool_calls ?? []) { + const toolName = call.function_name ?? "tool"; + if (!privateInfoRead(toolName, call.arguments)) continue; + const observation = step.observation?.results?.find( + (result) => result.source_call_id === call.tool_call_id, + ); + if (typeof observation?.content !== "string") continue; + for (const match of observation.content.matchAll( + /["']\s*:\s*["']([^"'\\]{4,})["']/g, + )) { + values.add(match[1]); + } + } + } + return [...values]; +} + +function phaseEvent( + rowId: string, + name: string, + phase: BenchmarkPhase, + metadata: Record = {}, +): BraintrustEvent | undefined { + if (phase.start === undefined || phase.end === undefined) return undefined; + const id = uuidV5(`${rowId}:phase:${name}`); + return { + id, + span_id: id, + root_span_id: rowId, + span_parents: [rowId], + span_attributes: { name, type: "task" }, + created: timestampIso(phase.start), + metadata: { phase: name, ...metadata }, + metrics: { + start: phase.start, + end: phase.end, + ...(phase.durationMs === undefined + ? {} + : { duration_ms: phase.durationMs }), + }, + _is_merge: false, + }; +} + +function phaseEvents(trial: BenchmarkTrial, rowId: string): BraintrustEvent[] { + const events: BraintrustEvent[] = []; + for (const [name, phase] of Object.entries({ + environment_setup: trial.phases.environmentSetup, + agent_setup: trial.phases.agentSetup, + agent_execution: trial.phases.agentExecution, + verifier: trial.phases.verifier, + })) { + if (!phase) continue; + const event = phaseEvent(rowId, name, phase); + if (event) events.push(event); + } + + const stepSetupStart = trial.phases.agentSetup?.end; + const stepSetupEnd = trial.phases.agentExecution?.start; + if ( + stepSetupStart !== undefined && + stepSetupEnd !== undefined && + stepSetupEnd > stepSetupStart + ) { + const event = phaseEvent(rowId, "step_setup", { + start: stepSetupStart, + end: stepSetupEnd, + durationMs: (stepSetupEnd - stepSetupStart) * 1000, + }); + if (event) events.push(event); + } + + if (trial.browser?.start !== undefined && trial.browser.end !== undefined) { + const event = phaseEvent( + rowId, + "browser_session", + { + start: trial.browser.start, + end: trial.browser.end, + durationMs: (trial.browser.end - trial.browser.start) * 1000, + }, + { + timeoutSeconds: trial.browser.timeoutSeconds, + deletionVerified: trial.browser.deletionVerified, + }, + ); + if (event) events.push(event); + } + + const finalizeStart = trial.phases.verifier?.end; + const finalizeEnd = trial.metrics.end; + if ( + finalizeStart !== undefined && + finalizeEnd !== undefined && + finalizeEnd > finalizeStart + ) { + const event = phaseEvent(rowId, "finalize", { + start: finalizeStart, + end: finalizeEnd, + durationMs: (finalizeEnd - finalizeStart) * 1000, + }); + if (event) events.push(event); + } + return events; +} + function atifEvents(trial: BenchmarkTrial, rowId: string): BraintrustEvent[] { const events: BraintrustEvent[] = []; - const agentSteps = trajectorySteps(trial).filter( - (candidate) => candidate.source === "agent", - ); - for (const [stepIndex, step] of agentSteps.entries()) { + const steps = trajectorySteps(trial); + const sensitiveValues = [ + ...collectSensitiveValues(steps), + ...privateInfoValues(steps), + ]; + const agentExecutionId = uuidV5(`${rowId}:phase:agent_execution`); + let previousEnd = trial.phases.agentExecution?.start; + + for (const [trajectoryIndex, step] of steps.entries()) { + if (step.source !== "agent") continue; const stepId = step.step_id; - const stepKey = `${stepId ?? "missing"}:${stepIndex}`; + const stepKey = `${stepId ?? "missing"}:${trajectoryIndex}`; const llmId = uuidV5(`${rowId}:llm:${stepKey}`); - const start = step.timestamp - ? Date.parse(step.timestamp) / 1000 - : undefined; + const end = timestampSeconds(step.timestamp); + const start = + previousEnd === undefined || end === undefined + ? end + : Math.min(previousEnd, end); + if (end !== undefined) previousEnd = end; + const promptTokens = number(step.metrics?.prompt_tokens); + const completionTokens = number(step.metrics?.completion_tokens); const llmMetrics = Object.fromEntries( Object.entries({ start, - end: start, - prompt_tokens: number(step.metrics?.prompt_tokens), - completion_tokens: number(step.metrics?.completion_tokens), + end, + prompt_tokens: promptTokens, + prompt_cached_tokens: number(step.metrics?.cached_tokens), + completion_tokens: completionTokens, + tokens: + promptTokens === undefined || completionTokens === undefined + ? undefined + : promptTokens + completionTokens, cost_usd: number(step.metrics?.cost_usd), }).filter((entry): entry is [string, number] => entry[1] !== undefined), ); @@ -186,14 +394,21 @@ function atifEvents(trial: BenchmarkTrial, rowId: string): BraintrustEvent[] { id: llmId, span_id: llmId, root_span_id: rowId, - span_parents: [rowId], + span_parents: [ + trial.phases.agentExecution?.start !== undefined && + trial.phases.agentExecution.end !== undefined + ? agentExecutionId + : rowId, + ], span_attributes: { name: "agent", type: "llm" }, - created: step.timestamp, - output: redactValue(step.message), + created: timestampIso(start) ?? step.timestamp, + input: llmInput(steps, trajectoryIndex, sensitiveValues), + output: llmOutput(step, sensitiveValues), metadata: { phase: "agent_execution", + timing: "ATIF turn completion interval", stepId, - stepIndex, + trajectoryIndex, model: step.model_name ?? trial.model, }, metrics: llmMetrics, @@ -207,22 +422,34 @@ function atifEvents(trial: BenchmarkTrial, rowId: string): BraintrustEvent[] { const observation = step.observation?.results?.find( (result) => result.source_call_id === call.tool_call_id, ); + const toolName = call.function_name ?? "tool"; + const sessionId = + call.arguments !== null && typeof call.arguments === "object" + ? (call.arguments as Record).session_id + : undefined; + const sessionIdMatchesExpected = + typeof sessionId === "string" && trial.expectedBrowserSessionId + ? sessionId === trial.expectedBrowserSessionId + : undefined; events.push({ id: toolId, span_id: toolId, root_span_id: rowId, span_parents: [llmId], - span_attributes: { name: call.function_name ?? "tool", type: "tool" }, + span_attributes: { name: toolName, type: "tool" }, created: step.timestamp, - input: redactValue(call.arguments), - output: redactValue(observation?.content), + input: redactValueWithSecrets(call.arguments, sensitiveValues), + output: privateInfoRead(toolName, call.arguments) + ? REDACTED_PRIVATE_INFO + : redactValueWithSecrets(observation?.content, sensitiveValues), metadata: { phase: "agent_execution", + timing: "not available in ATIF", stepId, - stepIndex, + trajectoryIndex, toolCallId: call.tool_call_id, + sessionIdMatchesExpected, }, - metrics: start === undefined ? undefined : { start: start, end: start }, _is_merge: false, }); } @@ -249,7 +476,11 @@ export function buildExperimentEvents( span_parents: [], span_attributes: { name: trial.taskName, type: "eval" }, created: trial.startedAt, - input: { source: trial.source, taskName: trial.taskName }, + input: { + source: trial.source, + taskName: trial.taskName, + instruction: taskInstruction(trial), + }, output: { reward: primaryReward?.value, rewardKey: primaryReward?.key, @@ -262,9 +493,11 @@ export function buildExperimentEvents( metrics: metricRecord(trial), _is_merge: false, }); + events.push(...phaseEvents(trial, rowId)); events.push(...atifEvents(trial, rowId)); } } + assertSafeToPublish(events); return events; } diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 851c7921..aa74415e 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -1,7 +1,46 @@ const SECRET_NAME = /(API_KEY|TOKEN|SECRET|PASSWORD|PRIVATE_KEY|CREDENTIAL)/i; -const SENSITIVE_FIELD = - /(API_KEY|TOKEN|JWT|SECRET|PASSWORD|PRIVATE_KEY|CREDENTIAL|^COOKIE$|^SET-COOKIE$)/i; const REDACTED = "[REDACTED]"; +const REDACTED_PRIVATE_INFO = "[REDACTED_PRIVATE_INFO]"; + +const SENSITIVE_FIELDS = new Set([ + "api_key", + "access_token", + "auth_token", + "refresh_token", + "session_token", + "token", + "jwt", + "secret", + "password", + "private_key", + "credential", + "credentials", + "cookie", + "set-cookie", + "session_id", + "replay_id", + "cdp_url", + "cdp_ws_url", + "viewer_url", + "browser_live_view_url", + "email", + "phone", + "address", +]); + +function normalizedField(key: string): string { + return key.trim().toLowerCase().replace(/-/g, "_"); +} + +function sensitiveField(key: string): boolean { + const normalized = normalizedField(key); + return ( + SENSITIVE_FIELDS.has(normalized) || + /(?:^|_)(?:api_key|access_token|auth_token|refresh_token|session_token|password|private_key|credential|session_id|replay_id|cdp_url|viewer_url)$/.test( + normalized, + ) + ); +} function secretValues(): string[] { return Object.entries(process.env) @@ -12,22 +51,34 @@ function secretValues(): string[] { .sort((left, right) => right.length - left.length); } -export function redactString(value: string, maxLength = 20_000): string { +function redactTypedLiterals(value: string): string { + return value.replace( + /(\.(?:fill|type)\(\s*)(["'`])(?:\\.|(?!\2)[^\\])*?\2/g, + (_match, prefix: string, quote: string) => + `${prefix}${quote}${REDACTED}${quote}`, + ); +} + +export function redactStringWithSecrets( + value: string, + additionalSecrets: string[], + maxLength = 20_000, +): string { let redacted = value; - for (const secret of secretValues()) { - redacted = redacted.split(secret).join(REDACTED); + for (const secret of [...secretValues(), ...additionalSecrets]) { + if (secret.length >= 4) redacted = redacted.split(secret).join(REDACTED); } - redacted = redacted + redacted = redactTypedLiterals(redacted) .replace(/\bBearer\s+[A-Za-z0-9._~+/=-]+/gi, `Bearer ${REDACTED}`) .replace(/\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\b/g, REDACTED) .replace(/\b(?:sk|pk|bt|kapi|whsec)[-_][A-Za-z0-9_-]{12,}\b/gi, REDACTED) .replace( - /(["']?(?:api[_-]?key|access[_-]?token|credential|jwt|password|secret|token)["']?\s*[:=]\s*["']?)[^"'\s,}&]+/gi, + /(["']?(?:api[_-]?key|access[_-]?token|auth[_-]?token|credential|jwt|password|private[_-]?key|refresh[_-]?token|replay[_-]?id|secret|session[_-]?id|session[_-]?token|token)["']?\s*[:=]\s*["']?)[^"'\s,}&]+/gi, `$1${REDACTED}`, ) .replace( - /([?&](?:api[_-]?key|access[_-]?token|auth|code|credential|jwt|password|secret|session[_-]?token|token)=)[^&#\s]+/gi, + /([?&](?:api[_-]?key|access[_-]?token|auth|code|credential|jwt|password|secret|session[_-]?id|session[_-]?token|token)=)[^&#\s]+/gi, `$1${REDACTED}`, ) .replace(/(\b(?:cookie|set-cookie)\s*:\s*)[^\r\n]+/gi, `$1${REDACTED}`) @@ -40,20 +91,138 @@ export function redactString(value: string, maxLength = 20_000): string { : redacted; } -export function redactValue(value: unknown, maxStringLength = 20_000): unknown { - if (typeof value === "string") return redactString(value, maxStringLength); +export function redactString(value: string, maxLength = 20_000): string { + return redactStringWithSecrets(value, [], maxLength); +} + +export function redactValueWithSecrets( + value: unknown, + additionalSecrets: string[], + maxStringLength = 20_000, +): unknown { + if (typeof value === "string") { + return redactStringWithSecrets(value, additionalSecrets, maxStringLength); + } if (Array.isArray(value)) { - return value.map((entry) => redactValue(entry, maxStringLength)); + return value.map((entry) => + redactValueWithSecrets(entry, additionalSecrets, maxStringLength), + ); } if (value !== null && typeof value === "object") { return Object.fromEntries( Object.entries(value as Record).map(([key, entry]) => [ key, - SENSITIVE_FIELD.test(key) + sensitiveField(key) ? REDACTED - : redactValue(entry, maxStringLength), + : redactValueWithSecrets(entry, additionalSecrets, maxStringLength), ]), ); } return value; } + +export function redactValue(value: unknown, maxStringLength = 20_000): unknown { + return redactValueWithSecrets(value, [], maxStringLength); +} + +export function collectSensitiveValues(value: unknown): string[] { + const values = new Set(); + const collectString = (text: string) => { + for (const match of text.matchAll( + /["']?(?:password|private[_-]?key|secret|session[_-]?id|replay[_-]?id)["']?\s*[:=]\s*["']([^"'\s,}&]{4,})/gi, + )) { + values.add(match[1]); + } + for (const match of text.matchAll( + /\.(?:fill|type)\(\s*(["'`])((?:\\.|(?!\1).){4,})\1/g, + )) { + values.add(match[2]); + } + for (const match of text.matchAll( + /["'](?:id|name|type)["']\s*:\s*["'][^"']*password[^"']*["'][\s\S]{0,300}?["']value["']\s*:\s*["']([^"']{4,})["']/gi, + )) { + values.add(match[1]); + } + }; + const visit = (entry: unknown, key?: string) => { + if (typeof entry === "string") { + if (key && sensitiveField(key) && entry.length >= 4) values.add(entry); + collectString(entry); + return; + } + if (Array.isArray(entry)) { + for (const item of entry) visit(item); + return; + } + if (entry !== null && typeof entry === "object") { + for (const [childKey, child] of Object.entries( + entry as Record, + )) { + visit(child, childKey); + } + } + }; + visit(value); + return [...values].sort((left, right) => right.length - left.length); +} + +export function privateInfoRead(toolName: string, input: unknown): boolean { + if (!/(?:^|__)exec_command$/.test(toolName)) return false; + const command = + input !== null && typeof input === "object" + ? String((input as Record).cmd ?? "") + : String(input ?? ""); + return /(?:^|\s|["'])\/?(?:workspace\/)?my-info\/(?!kernel_browser\.json)/.test( + command, + ); +} + +function assertSafeString(value: string): void { + if (/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/i.test(value)) { + throw new Error("Braintrust payload still contains an email address"); + } + for (const match of value.matchAll( + /\.(?:fill|type)\(\s*(["'`])((?:\\.|(?!\1).)*)\1/g, + )) { + if (match[2] !== REDACTED) { + throw new Error("Braintrust payload still contains a typed form value"); + } + } + for (const match of value.matchAll( + /["']?(?:api[_-]?key|access[_-]?token|auth[_-]?token|credential|jwt|password|private[_-]?key|refresh[_-]?token|replay[_-]?id|secret|session[_-]?id|session[_-]?token)["']?\s*[:=]\s*["']?([^"'\s,}&]+)/gi, + )) { + if (match[1] !== REDACTED) { + throw new Error( + "Braintrust payload still contains a sensitive field value", + ); + } + } + for (const secret of secretValues()) { + if (value.includes(secret)) { + throw new Error("Braintrust payload still contains a configured secret"); + } + } +} + +export function assertSafeToPublish(value: unknown): void { + if (typeof value === "string") { + assertSafeString(value); + return; + } + if (Array.isArray(value)) { + for (const entry of value) assertSafeToPublish(entry); + return; + } + if (value !== null && typeof value === "object") { + for (const [key, entry] of Object.entries( + value as Record, + )) { + if (sensitiveField(key) && entry !== REDACTED) { + throw new Error(`Braintrust payload did not redact ${key}`); + } + assertSafeToPublish(entry); + } + } +} + +export { REDACTED, REDACTED_PRIVATE_INFO }; diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index a0774091..4a0911d3 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -11,7 +11,12 @@ import { join } from "node:path"; import { buildExperimentEvents, publishBenchmark } from "./publish-braintrust"; import { renderMarkdown } from "./report"; import { readBenchmarkArm, selectPrimaryReward, summarizeArm } from "./results"; -import { redactString, redactValue } from "./redact"; +import { + assertSafeToPublish, + privateInfoRead, + redactString, + redactValue, +} from "./redact"; import { assertProjectScopedCredential } from "./verify-project-scope"; const temporaryDirectories: string[] = []; @@ -66,6 +71,14 @@ function fixture(): string { }, started_at: "2026-01-01T00:00:00Z", finished_at: "2026-01-01T00:01:00Z", + environment_setup: { + started_at: "2026-01-01T00:00:00Z", + finished_at: "2026-01-01T00:00:05Z", + }, + agent_setup: { + started_at: "2026-01-01T00:00:05Z", + finished_at: "2026-01-01T00:00:10Z", + }, step_results: [ { agent_result: { @@ -74,6 +87,14 @@ function fixture(): string { n_output_tokens: 20, cost_usd: 0.01, }, + agent_execution: { + started_at: "2026-01-01T00:00:20Z", + finished_at: "2026-01-01T00:00:40Z", + }, + verifier: { + started_at: "2026-01-01T00:00:45Z", + finished_at: "2026-01-01T00:00:55Z", + }, }, ], }); @@ -81,19 +102,40 @@ function fixture(): string { steps: [ { step_id: 1, + source: "system", + timestamp: "2026-01-01T00:00:20Z", + message: "system prompt", + }, + { + step_id: 2, + source: "user", + timestamp: "2026-01-01T00:00:20Z", + message: "perform the task", + }, + { + step_id: 3, source: "agent", - timestamp: "2026-01-01T00:00:01Z", - message: "working", + timestamp: "2026-01-01T00:00:21Z", + message: "", tool_calls: [ { tool_call_id: "call-1", function_name: "execute_playwright_code", - arguments: { code: "return 'done'" }, + arguments: { + session_id: "session-123", + code: "await page.locator('#password').fill('secret-password'); return 'done'", + }, }, ], observation: { results: [{ source_call_id: "call-1", content: "done" }], }, + metrics: { + prompt_tokens: 100, + cached_tokens: 80, + completion_tokens: 20, + cost_usd: 0.01, + }, }, ], }); @@ -101,6 +143,20 @@ function fixture(): string { kernel_mcp_server_sha: "server-sha", clawbench_source_sha: "clawbench-sha", }); + writeJson(join(success, "steps/run/verifier/kernel-mcp-result.json"), { + expected_session_id: "session-123", + }); + writeJson( + join(success, "steps/run/verifier/data/kernel-browser-lifecycle.json"), + { + timeout_seconds: 1920, + deletion_verified: true, + events: [ + { event: "browser_created", ts: 1767225620 }, + { event: "browser_deleted", ts: 1767225640 }, + ], + }, + ); const failed = join(root, "task-two__def"); writeJson(join(failed, "result.json"), { @@ -154,6 +210,30 @@ describe("Harbor result ingestion", () => { }); }); + test("classifies ungraded step setup failures as infrastructure", () => { + const root = fixture(); + const failedPath = join(root, "task-two__def", "result.json"); + const failed = JSON.parse(readFileSync(failedPath, "utf8")) as Record< + string, + unknown + >; + delete failed.exception_info; + failed.verifier_result = null; + failed.step_results = [ + { + exception_info: { + exception_type: "RuntimeError", + exception_message: "Step setup exited with code 1", + }, + }, + ]; + writeJson(failedPath, failed); + + const arm = readBenchmarkArm({ name: "candidate", path: root }); + expect(arm.trials[1].errorClass).toBe("infra"); + expect(arm.trials[1].error).toContain("Step setup exited with code 1"); + }); + test("summarizes against the intended task denominator", () => { const summary = summarizeArm( readBenchmarkArm({ name: "candidate", path: fixture() }), @@ -207,10 +287,14 @@ describe("Harbor result ingestion", () => { expect( first.filter((event) => event.span_attributes.type === "tool"), ).toHaveLength(1); + expect( + first.filter((event) => event.span_attributes.type === "task"), + ).toHaveLength(7); const root = first.find((event) => event.span_attributes.type === "eval"); expect(root?.input).toEqual({ source: "clawbench-v2", taskName: "v2-task-one", + instruction: "perform the task", }); expect(root?.span_parents).toEqual([]); const infra = first.find( @@ -229,6 +313,55 @@ describe("Harbor result ingestion", () => { reward: 1, rewardKey: "reward_lenient", }); + const llm = first.find((event) => event.span_attributes.type === "llm"); + expect(llm?.input).toEqual([ + { + source: "system", + message: "system prompt", + toolCalls: [], + observations: [], + }, + { + source: "user", + message: "perform the task", + toolCalls: [], + observations: [], + }, + ]); + expect(llm?.output).toMatchObject({ + message: "", + toolCalls: [ + { + name: "execute_playwright_code", + arguments: { + session_id: "[REDACTED]", + code: "await page.locator('#password').fill('[REDACTED]'); return 'done'", + }, + }, + ], + }); + const tool = first.find((event) => event.span_attributes.type === "tool"); + expect(tool?.metadata).toMatchObject({ + sessionIdMatchesExpected: true, + }); + const browser = first.find( + (event) => event.span_attributes.name === "browser_session", + ); + expect(browser?.metadata).toMatchObject({ + timeoutSeconds: 1920, + deletionVerified: true, + }); + expect(llm?.metrics).toMatchObject({ + start: Date.parse("2026-01-01T00:00:20Z") / 1000, + end: Date.parse("2026-01-01T00:00:21Z") / 1000, + prompt_tokens: 100, + prompt_cached_tokens: 80, + completion_tokens: 20, + tokens: 120, + cost_usd: 0.01, + }); + expect(root?.metrics).not.toHaveProperty("input_tokens"); + expect(root?.metrics).not.toHaveProperty("cost_usd"); }); test("re-publishes the same rows and spans by deterministic ID", async () => { @@ -409,22 +542,39 @@ describe("Braintrust redaction", () => { process.env.TEST_API_KEY = "super-secret-value"; expect( redactString( - 'Bearer super-secret-value sk-proj-abcdefghijklmnop?access_token=visible&token=plain&jwt=opaque "password":"generated-password" Cookie: session=visible\nhttps://example.com/browser/live/replay-slug user@example.com', + 'Bearer super-secret-value sk-proj-abcdefghijklmnop?access_token=visible&token=plain&jwt=opaque "password":"generated-password" "session_id":"session-123" Cookie: session=visible\nhttps://example.com/browser/live/replay-slug user@example.com await page.locator("#password").fill("typed-password")', ), ).toBe( - 'Bearer [REDACTED] [REDACTED]?access_token=[REDACTED]&token=[REDACTED]&jwt=[REDACTED] "password":"[REDACTED]" Cookie: [REDACTED]\nhttps://example.com/browser/live/[REDACTED] [REDACTED_EMAIL]', + 'Bearer [REDACTED] [REDACTED]?access_token=[REDACTED]&token=[REDACTED]&jwt=[REDACTED] "password":"[REDACTED]" "session_id":"[REDACTED]" Cookie: [REDACTED]\nhttps://example.com/browser/live/[REDACTED] [REDACTED_EMAIL] await page.locator("#password").fill("[REDACTED]")', ); - expect( - redactValue({ - api_key: "visible", - Cookie: "session=visible", - nested: ["bt-abcdefghijklmnop"], - }), - ).toEqual({ + const redacted = redactValue({ + api_key: "visible", + Cookie: "session=visible", + session_id: "session-123", + max_output_tokens: 1000, + nested: ["bt-abcdefghijklmnop"], + }); + expect(redacted).toEqual({ api_key: "[REDACTED]", Cookie: "[REDACTED]", + session_id: "[REDACTED]", + max_output_tokens: 1000, nested: ["[REDACTED]"], }); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + expect(() => + assertSafeToPublish({ code: "page.fill('still-visible')" }), + ).toThrow("typed form value"); + expect( + privateInfoRead("exec_command", { + cmd: "cat /my-info/email_credentials.json", + }), + ).toBe(true); + expect( + privateInfoRead("exec_command", { + cmd: "cat /my-info/kernel_browser.json", + }), + ).toBe(false); delete process.env.TEST_API_KEY; }); }); @@ -473,6 +623,7 @@ describe("benchmark workflow hardening", () => { "harbor_hypeman_version=${HARBOR_HYPEMAN_VERSION:-0.1.2}", ); expect(runner).toContain('--max-retries "${HARBOR_MAX_RETRIES:-5}"'); + expect(runner).toContain('bun "$benchmark_dir/verify-purelymail.ts"'); for (const exception of [ "APITimeoutError", "APIConnectionError", @@ -480,6 +631,8 @@ describe("benchmark workflow hardening", () => { "InternalServerError", "ConnectionRefusedError", "ExecProtocolError", + "AgentSetupTimeoutError", + "RuntimeError", ]) { expect(runner).toContain(`--retry-include ${exception}`); } diff --git a/benchmarks/harbor/results.ts b/benchmarks/harbor/results.ts index 41def571..c2024a05 100644 --- a/benchmarks/harbor/results.ts +++ b/benchmarks/harbor/results.ts @@ -20,6 +20,21 @@ export interface BenchmarkMetrics { end?: number; } +export interface BenchmarkPhase { + startedAt?: string; + finishedAt?: string; + start?: number; + end?: number; + durationMs?: number; +} + +export interface BenchmarkBrowserLifecycle { + timeoutSeconds?: number; + start?: number; + end?: number; + deletionVerified?: boolean; +} + export interface BenchmarkTrial { arm: string; id: string; @@ -35,6 +50,14 @@ export interface BenchmarkTrial { error?: string; errorClass?: "infra"; metrics: BenchmarkMetrics; + phases: { + environmentSetup?: BenchmarkPhase; + agentSetup?: BenchmarkPhase; + agentExecution?: BenchmarkPhase; + verifier?: BenchmarkPhase; + }; + browser?: BenchmarkBrowserLifecycle; + expectedBrowserSessionId?: string; kernelMcpSha?: string; clawbenchSha?: string; trajectoryPath?: string; @@ -154,6 +177,20 @@ function durationMs( return Number.isFinite(duration) && duration >= 0 ? duration : undefined; } +function phase(value: unknown): BenchmarkPhase | undefined { + const timing = object(value); + const startedAt = string(timing.started_at); + const finishedAt = string(timing.finished_at); + if (!startedAt && !finishedAt) return undefined; + return { + startedAt, + finishedAt, + start: isoSeconds(startedAt), + end: isoSeconds(finishedAt), + durationMs: durationMs(startedAt, finishedAt), + }; +} + function sumStepMetric( stepResults: unknown[], key: string, @@ -224,21 +261,40 @@ function parseTrial(arm: string, trialDir: string): BenchmarkTrial { const agentInfo = object(result.agent_info); const modelInfo = object(agentInfo.model_info); const steps = array(result.step_results); - const exception = result.exception_info; + const rewards = trialRewards(result, trialDir); const exceptionFile = join(trialDir, "exception.txt"); + const stepException = + Object.keys(rewards).length === 0 + ? steps.map((step) => object(step).exception_info).find(Boolean) + : undefined; const error = errorText( - exception ?? + result.exception_info ?? + stepException ?? (existsSync(exceptionFile) ? readFileSync(exceptionFile, "utf8") : undefined), ); - const rewards = trialRewards(result, trialDir); const startedAt = string(result.started_at); const finishedAt = string(result.finished_at); const trajectoryPath = join(trialDir, "steps/run/agent/trajectory.json"); const runManifest = readJsonIfPresent( join(trialDir, "steps/run/verifier/kernel-mcp/run-manifest.json"), ); + const kernelMcpResult = readJsonIfPresent( + join(trialDir, "steps/run/verifier/kernel-mcp-result.json"), + ); + const browserLifecycle = readJsonIfPresent( + join(trialDir, "steps/run/verifier/data/kernel-browser-lifecycle.json"), + ); + const browserEvents = array(browserLifecycle.events).map(object); + const browserCreated = browserEvents.find( + (event) => event.event === "browser_created", + ); + const browserDeleted = browserEvents.find( + (event) => event.event === "browser_deleted", + ); + const browserStart = number(browserCreated?.ts); + const browserEnd = number(browserDeleted?.ts); return { arm, @@ -274,6 +330,25 @@ function parseTrial(arm: string, trialDir: string): BenchmarkTrial { end: isoSeconds(finishedAt), ...trajectoryMetrics(trajectoryPath), }, + phases: { + environmentSetup: phase(result.environment_setup), + agentSetup: phase(result.agent_setup), + agentExecution: phase(object(steps.at(-1)).agent_execution), + verifier: phase(object(steps.at(-1)).verifier), + }, + browser: + browserStart === undefined && browserEnd === undefined + ? undefined + : { + timeoutSeconds: number(browserLifecycle.timeout_seconds), + start: browserStart, + end: browserEnd, + deletionVerified: + typeof browserLifecycle.deletion_verified === "boolean" + ? browserLifecycle.deletion_verified + : undefined, + }, + expectedBrowserSessionId: string(kernelMcpResult.expected_session_id), kernelMcpSha: string(runManifest.kernel_mcp_server_sha), clawbenchSha: string(runManifest.clawbench_source_sha), trajectoryPath: existsSync(trajectoryPath) ? trajectoryPath : undefined, diff --git a/benchmarks/harbor/verify-purelymail.test.ts b/benchmarks/harbor/verify-purelymail.test.ts new file mode 100644 index 00000000..d49bcd62 --- /dev/null +++ b/benchmarks/harbor/verify-purelymail.test.ts @@ -0,0 +1,46 @@ +import { describe, expect, test } from "bun:test"; +import { verifyPurelyMail } from "./verify-purelymail"; + +describe("PurelyMail benchmark preflight", () => { + test("creates and deletes a disposable account", async () => { + const requests: Array<{ endpoint: string; body: Record }> = + []; + const fetcher = (async (request, init) => { + requests.push({ + endpoint: new URL(String(request)).pathname.split("/").at(-1) ?? "", + body: JSON.parse(String(init?.body)), + }); + return Response.json({ type: "success" }); + }) as typeof fetch; + + await verifyPurelyMail("api-key", "example.test", fetcher); + + expect(requests.map((request) => request.endpoint)).toEqual([ + "createUser", + "deleteUser", + ]); + expect(requests[0].body).toMatchObject({ + domainName: "example.test", + enablePasswordReset: false, + sendWelcomeEmail: false, + }); + expect(requests[1].body.userName).toBe( + `${requests[0].body.userName}@example.test`, + ); + }); + + test("fails closed on API errors without printing credentials", async () => { + const fetcher = (async () => + Response.json({ + type: "error", + code: "invalidToken", + message: "Token not valid.", + })) as unknown as typeof fetch; + + await expect( + verifyPurelyMail("secret-api-key", "example.test", fetcher), + ).rejects.toThrow( + "PurelyMail createUser failed (invalidToken): Token not valid.", + ); + }); +}); diff --git a/benchmarks/harbor/verify-purelymail.ts b/benchmarks/harbor/verify-purelymail.ts new file mode 100644 index 00000000..8c8204fa --- /dev/null +++ b/benchmarks/harbor/verify-purelymail.ts @@ -0,0 +1,80 @@ +#!/usr/bin/env bun +import { randomBytes, randomUUID } from "node:crypto"; + +interface PurelyMailResponse { + type?: string; + code?: string; + message?: string; +} + +function env(name: string): string { + const value = process.env[name]?.trim(); + if (!value) throw new Error(`${name} is required`); + return value; +} + +async function request( + fetcher: typeof fetch, + apiKey: string, + endpoint: string, + body: Record, +): Promise { + const response = await fetcher(`https://purelymail.com/api/v0/${endpoint}`, { + method: "POST", + headers: { + "Purelymail-Api-Token": apiKey, + "Content-Type": "application/json", + }, + body: JSON.stringify(body), + }); + if (!response.ok) { + throw new Error(`PurelyMail ${endpoint} returned HTTP ${response.status}`); + } + const result = (await response.json()) as PurelyMailResponse; + if (!result || typeof result !== "object") { + throw new Error(`PurelyMail ${endpoint} returned an invalid response`); + } + if (result.type === "error") { + const code = result.code ? ` (${result.code})` : ""; + const message = result.message ? `: ${result.message}` : ""; + throw new Error(`PurelyMail ${endpoint} failed${code}${message}`); + } + return result; +} + +export async function verifyPurelyMail( + apiKey: string, + domain: string, + fetcher: typeof fetch = fetch, +): Promise { + const local = `cbpreflight${randomUUID().replace(/-/g, "").slice(0, 12)}`; + const email = `${local}@${domain}`; + const password = randomBytes(18).toString("base64url"); + let created = false; + try { + await request(fetcher, apiKey, "createUser", { + userName: local, + domainName: domain, + password, + enablePasswordReset: false, + sendWelcomeEmail: false, + }); + created = true; + } finally { + if (created) { + await request(fetcher, apiKey, "deleteUser", { userName: email }); + } + } +} + +async function main(): Promise { + await verifyPurelyMail(env("PURELY_MAIL_API_KEY"), env("PURELY_MAIL_DOMAIN")); + process.stdout.write("PurelyMail preflight passed\n"); +} + +if (import.meta.main) { + main().catch((error) => { + console.error(error instanceof Error ? error.message : error); + process.exitCode = 1; + }); +} From 2586a7041801211cdb3e2ab0010c9d2071dce052 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 2 Sep 2026 01:59:15 +0000 Subject: [PATCH 02/14] Cover alternate private input shapes --- benchmarks/harbor/redact.ts | 68 ++++++++++++++++++++++--------- benchmarks/harbor/results.test.ts | 19 ++++++++- 2 files changed, 66 insertions(+), 21 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index aa74415e..06f73a29 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -51,12 +51,30 @@ function secretValues(): string[] { .sort((left, right) => right.length - left.length); } +const TYPED_CALL = /\.(?:fill|type)\(([^)]*)\)/g; +const STRING_LITERAL = /(["'`])((?:\\.|(?!\1).)*)\1/g; + +function typedCallValues(value: string): string[] { + const values: string[] = []; + for (const call of value.matchAll(TYPED_CALL)) { + const literals = [...call[1].matchAll(STRING_LITERAL)]; + const typedValue = literals.at(-1)?.[2]; + if (typedValue !== undefined) values.push(typedValue); + } + return values; +} + function redactTypedLiterals(value: string): string { - return value.replace( - /(\.(?:fill|type)\(\s*)(["'`])(?:\\.|(?!\2)[^\\])*?\2/g, - (_match, prefix: string, quote: string) => - `${prefix}${quote}${REDACTED}${quote}`, - ); + return value.replace(TYPED_CALL, (call, argumentsText: string) => { + const literals = [...argumentsText.matchAll(STRING_LITERAL)]; + const typedValue = literals.at(-1); + if (!typedValue || typedValue.index === undefined) return call; + const start = typedValue.index; + const end = start + typedValue[0].length; + const quote = typedValue[1]; + const redactedArguments = `${argumentsText.slice(0, start)}${quote}${REDACTED}${quote}${argumentsText.slice(end)}`; + return call.replace(argumentsText, redactedArguments); + }); } export function redactStringWithSecrets( @@ -133,10 +151,8 @@ export function collectSensitiveValues(value: unknown): string[] { )) { values.add(match[1]); } - for (const match of text.matchAll( - /\.(?:fill|type)\(\s*(["'`])((?:\\.|(?!\1).){4,})\1/g, - )) { - values.add(match[2]); + for (const typedValue of typedCallValues(text)) { + if (typedValue.length >= 4) values.add(typedValue); } for (const match of text.matchAll( /["'](?:id|name|type)["']\s*:\s*["'][^"']*password[^"']*["'][\s\S]{0,300}?["']value["']\s*:\s*["']([^"']{4,})["']/gi, @@ -167,13 +183,29 @@ export function collectSensitiveValues(value: unknown): string[] { } export function privateInfoRead(toolName: string, input: unknown): boolean { - if (!/(?:^|__)exec_command$/.test(toolName)) return false; - const command = + if (!/(?:^|__)(?:exec_command|bash|read)$/i.test(toolName)) return false; + const fields = input !== null && typeof input === "object" - ? String((input as Record).cmd ?? "") - : String(input ?? ""); - return /(?:^|\s|["'])\/?(?:workspace\/)?my-info\/(?!kernel_browser\.json)/.test( - command, + ? (input as Record) + : {}; + const text = [ + fields.cmd, + fields.command, + fields.file_path, + fields.path, + typeof input === "string" ? input : undefined, + ] + .filter((entry) => entry !== undefined) + .map(String) + .join("\n"); + const paths = [ + ...text.matchAll( + /(?:^|[\s"'`=])(?:\.\/|\/(?:workspace\/)?|workspace\/)?my-info\/([^\s"'`;)]*)/g, + ), + ].map((match) => match[1]); + return paths.some( + (path) => + path.length === 0 || !/^kernel_browser\.json(?:$|[?#])/.test(path), ); } @@ -181,10 +213,8 @@ function assertSafeString(value: string): void { if (/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/i.test(value)) { throw new Error("Braintrust payload still contains an email address"); } - for (const match of value.matchAll( - /\.(?:fill|type)\(\s*(["'`])((?:\\.|(?!\1).)*)\1/g, - )) { - if (match[2] !== REDACTED) { + for (const typedValue of typedCallValues(value)) { + if (typedValue !== REDACTED) { throw new Error("Braintrust payload still contains a typed form value"); } } diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 4a0911d3..d2592236 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -542,10 +542,10 @@ describe("Braintrust redaction", () => { process.env.TEST_API_KEY = "super-secret-value"; expect( redactString( - 'Bearer super-secret-value sk-proj-abcdefghijklmnop?access_token=visible&token=plain&jwt=opaque "password":"generated-password" "session_id":"session-123" Cookie: session=visible\nhttps://example.com/browser/live/replay-slug user@example.com await page.locator("#password").fill("typed-password")', + 'Bearer super-secret-value sk-proj-abcdefghijklmnop?access_token=visible&token=plain&jwt=opaque "password":"generated-password" "session_id":"session-123" Cookie: session=visible\nhttps://example.com/browser/live/replay-slug user@example.com await page.locator("#password").fill("typed-password"); await page.fill("#password", "two-arg-secret")', ), ).toBe( - 'Bearer [REDACTED] [REDACTED]?access_token=[REDACTED]&token=[REDACTED]&jwt=[REDACTED] "password":"[REDACTED]" "session_id":"[REDACTED]" Cookie: [REDACTED]\nhttps://example.com/browser/live/[REDACTED] [REDACTED_EMAIL] await page.locator("#password").fill("[REDACTED]")', + 'Bearer [REDACTED] [REDACTED]?access_token=[REDACTED]&token=[REDACTED]&jwt=[REDACTED] "password":"[REDACTED]" "session_id":"[REDACTED]" Cookie: [REDACTED]\nhttps://example.com/browser/live/[REDACTED] [REDACTED_EMAIL] await page.locator("#password").fill("[REDACTED]"); await page.fill("#password", "[REDACTED]")', ); const redacted = redactValue({ api_key: "visible", @@ -565,11 +565,26 @@ describe("Braintrust redaction", () => { expect(() => assertSafeToPublish({ code: "page.fill('still-visible')" }), ).toThrow("typed form value"); + expect(() => + assertSafeToPublish({ + code: "page.fill('#password', 'still-visible')", + }), + ).toThrow("typed form value"); expect( privateInfoRead("exec_command", { cmd: "cat /my-info/email_credentials.json", }), ).toBe(true); + expect( + privateInfoRead("Bash", { + command: "cat ./my-info/alex_green_personal_info.json", + }), + ).toBe(true); + expect( + privateInfoRead("Read", { + file_path: "/workspace/my-info/email_credentials.json", + }), + ).toBe(true); expect( privateInfoRead("exec_command", { cmd: "cat /my-info/kernel_browser.json", From 8f7a8f224cb0c6263decc5d9024c422125915360 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 30 Sep 2026 10:48:40 +0000 Subject: [PATCH 03/14] Address benchmark trace review --- benchmarks/harbor/publish-braintrust.ts | 12 +- benchmarks/harbor/redact.ts | 187 +++++++++++++++++++++--- benchmarks/harbor/results.test.ts | 68 ++++++++- 3 files changed, 241 insertions(+), 26 deletions(-) diff --git a/benchmarks/harbor/publish-braintrust.ts b/benchmarks/harbor/publish-braintrust.ts index 4a8ad0d6..cb42ac90 100644 --- a/benchmarks/harbor/publish-braintrust.ts +++ b/benchmarks/harbor/publish-braintrust.ts @@ -132,12 +132,22 @@ function trajectorySteps(trial: BenchmarkTrial): AtifStep[] { } function metricRecord(trial: BenchmarkTrial): Record { + const promptTokens = trial.metrics.inputTokens; + const completionTokens = trial.metrics.outputTokens; return Object.fromEntries( Object.entries({ start: trial.metrics.start, end: trial.metrics.end, duration_ms: trial.metrics.durationMs, tool_calls: trial.metrics.toolCalls, + prompt_tokens: promptTokens, + prompt_cached_tokens: trial.metrics.cacheTokens, + completion_tokens: completionTokens, + tokens: + promptTokens === undefined || completionTokens === undefined + ? undefined + : promptTokens + completionTokens, + estimated_cost: trial.metrics.costUsd, }).filter((entry): entry is [string, number] => entry[1] !== undefined), ); } @@ -387,7 +397,7 @@ function atifEvents(trial: BenchmarkTrial, rowId: string): BraintrustEvent[] { promptTokens === undefined || completionTokens === undefined ? undefined : promptTokens + completionTokens, - cost_usd: number(step.metrics?.cost_usd), + estimated_cost: number(step.metrics?.cost_usd), }).filter((entry): entry is [string, number] => entry[1] !== undefined), ); events.push({ diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 06f73a29..9ca20544 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -51,30 +51,103 @@ function secretValues(): string[] { .sort((left, right) => right.length - left.length); } -const TYPED_CALL = /\.(?:fill|type)\(([^)]*)\)/g; -const STRING_LITERAL = /(["'`])((?:\\.|(?!\1).)*)\1/g; +const TYPING_METHODS = new Set([ + "fill", + "type", + "pressSequentially", + "insertText", +]); -function typedCallValues(value: string): string[] { - const values: string[] = []; - for (const call of value.matchAll(TYPED_CALL)) { - const literals = [...call[1].matchAll(STRING_LITERAL)]; - const typedValue = literals.at(-1)?.[2]; - if (typedValue !== undefined) values.push(typedValue); +interface TypedLiteral { + start: number; + end: number; +} + +function typedLiterals(value: string): TypedLiteral[] { + const callStart = /^\.([A-Za-z]+)\s*\(/; + const literals: TypedLiteral[] = []; + let cursor = 0; + + while (cursor < value.length) { + const quote = value[cursor]; + if (quote === '"' || quote === "'" || quote === "`") { + cursor += 1; + while (cursor < value.length) { + if (value[cursor] === "\\") { + cursor += 2; + } else if (value[cursor] === quote) { + cursor += 1; + break; + } else { + cursor += 1; + } + } + continue; + } + + const match = value.slice(cursor).match(callStart); + if (!match || !TYPING_METHODS.has(match[1])) { + cursor += 1; + continue; + } + + let callCursor = cursor + match[0].length; + let depth = 1; + let lastLiteral: TypedLiteral | undefined; + while (callCursor < value.length && depth > 0) { + const callQuote = value[callCursor]; + if (callQuote === '"' || callQuote === "'" || callQuote === "`") { + const start = callCursor; + callCursor += 1; + while (callCursor < value.length) { + if (value[callCursor] === "\\") { + callCursor += 2; + } else if (value[callCursor] === callQuote) { + callCursor += 1; + break; + } else { + callCursor += 1; + } + } + lastLiteral = { start, end: callCursor }; + continue; + } + if (value.startsWith("//", callCursor)) { + const newline = value.indexOf("\n", callCursor + 2); + callCursor = newline === -1 ? value.length : newline + 1; + continue; + } + if (value.startsWith("/*", callCursor)) { + const commentEnd = value.indexOf("*/", callCursor + 2); + callCursor = commentEnd === -1 ? value.length : commentEnd + 2; + continue; + } + if (value[callCursor] === "(") depth += 1; + if (value[callCursor] === ")") depth -= 1; + callCursor += 1; + } + + if (lastLiteral) literals.push(lastLiteral); + cursor = callCursor; } - return values; + return literals; +} + +function typedCallValues(value: string): string[] { + return typedLiterals(value).map((literal) => + value.slice(literal.start + 1, literal.end - 1), + ); } function redactTypedLiterals(value: string): string { - return value.replace(TYPED_CALL, (call, argumentsText: string) => { - const literals = [...argumentsText.matchAll(STRING_LITERAL)]; - const typedValue = literals.at(-1); - if (!typedValue || typedValue.index === undefined) return call; - const start = typedValue.index; - const end = start + typedValue[0].length; - const quote = typedValue[1]; - const redactedArguments = `${argumentsText.slice(0, start)}${quote}${REDACTED}${quote}${argumentsText.slice(end)}`; - return call.replace(argumentsText, redactedArguments); - }); + let redacted = value; + for (const literal of typedLiterals(value).sort( + (left, right) => right.start - left.start, + )) { + const quote = value[literal.start]; + redacted = `${redacted.slice(0, literal.start)}${quote}${REDACTED}${quote}${redacted.slice(literal.end)}`; + } + return redacted; } export function redactStringWithSecrets( @@ -209,15 +282,81 @@ export function privateInfoRead(toolName: string, input: unknown): boolean { ); } +function assertTypedCallsRedacted(value: string): void { + const callStart = /^\.(?:fill|type|pressSequentially|insertText)\s*\(/; + let cursor = 0; + + while (cursor < value.length) { + const quote = value[cursor]; + if (quote === '"' || quote === "'" || quote === "`") { + cursor += 1; + while (cursor < value.length) { + if (value[cursor] === "\\") { + cursor += 2; + } else if (value[cursor] === quote) { + cursor += 1; + break; + } else { + cursor += 1; + } + } + continue; + } + + const match = value.slice(cursor).match(callStart); + if (!match) { + cursor += 1; + continue; + } + + let callCursor = cursor + match[0].length; + let depth = 1; + let lastLiteral: string | undefined; + while (callCursor < value.length && depth > 0) { + const callQuote = value[callCursor]; + if (callQuote === '"' || callQuote === "'" || callQuote === "`") { + const literalStart = callCursor + 1; + callCursor += 1; + while (callCursor < value.length) { + if (value[callCursor] === "\\") { + callCursor += 2; + } else if (value[callCursor] === callQuote) { + break; + } else { + callCursor += 1; + } + } + lastLiteral = value.slice(literalStart, callCursor); + if (callCursor < value.length) callCursor += 1; + continue; + } + if (value.startsWith("//", callCursor)) { + const newline = value.indexOf("\n", callCursor + 2); + callCursor = newline === -1 ? value.length : newline + 1; + continue; + } + if (value.startsWith("/*", callCursor)) { + const commentEnd = value.indexOf("*/", callCursor + 2); + callCursor = commentEnd === -1 ? value.length : commentEnd + 2; + continue; + } + if (value[callCursor] === "(") depth += 1; + if (value[callCursor] === ")") depth -= 1; + callCursor += 1; + } + + if (lastLiteral !== undefined && lastLiteral !== REDACTED) { + throw new Error("Braintrust payload still contains a typed form value"); + } + cursor = callCursor; + } +} + function assertSafeString(value: string): void { if (/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/i.test(value)) { throw new Error("Braintrust payload still contains an email address"); } - for (const typedValue of typedCallValues(value)) { - if (typedValue !== REDACTED) { - throw new Error("Braintrust payload still contains a typed form value"); - } - } + assertTypedCallsRedacted(value); for (const match of value.matchAll( /["']?(?:api[_-]?key|access[_-]?token|auth[_-]?token|credential|jwt|password|private[_-]?key|refresh[_-]?token|replay[_-]?id|secret|session[_-]?id|session[_-]?token)["']?\s*[:=]\s*["']?([^"'\s,}&]+)/gi, )) { diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index d2592236..b7eb8b50 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -13,6 +13,7 @@ import { renderMarkdown } from "./report"; import { readBenchmarkArm, selectPrimaryReward, summarizeArm } from "./results"; import { assertSafeToPublish, + collectSensitiveValues, privateInfoRead, redactString, redactValue, @@ -358,12 +359,42 @@ describe("Harbor result ingestion", () => { prompt_cached_tokens: 80, completion_tokens: 20, tokens: 120, - cost_usd: 0.01, + estimated_cost: 0.01, + }); + expect(root?.metrics).toMatchObject({ + prompt_tokens: 100, + prompt_cached_tokens: 80, + completion_tokens: 20, + tokens: 120, + estimated_cost: 0.01, }); expect(root?.metrics).not.toHaveProperty("input_tokens"); expect(root?.metrics).not.toHaveProperty("cost_usd"); }); + test("retains row cost when ATIF has no per-turn cost", () => { + const root = fixture(); + const trajectoryPath = join( + root, + "task-one__abc", + "steps/run/agent/trajectory.json", + ); + const trajectory = JSON.parse(readFileSync(trajectoryPath, "utf8")) as { + steps: Array<{ metrics?: Record }>; + }; + delete trajectory.steps[2].metrics?.cost_usd; + writeJson(trajectoryPath, trajectory); + + const events = buildExperimentEvents( + [readBenchmarkArm({ name: "candidate", path: root })], + "row-cost-fallback", + ); + const row = events.find((event) => event.span_attributes.type === "eval"); + const llm = events.find((event) => event.span_attributes.type === "llm"); + expect(row?.metrics).toMatchObject({ estimated_cost: 0.01 }); + expect(llm?.metrics).not.toHaveProperty("estimated_cost"); + }); + test("re-publishes the same rows and spans by deterministic ID", async () => { const arm = completeArm(); const originalFetch = globalThis.fetch; @@ -592,6 +623,41 @@ describe("Braintrust redaction", () => { ).toBe(false); delete process.env.TEST_API_KEY; }); + + test("redacts complex Playwright typing calls and rejects originals", () => { + const calls = [ + `page.fill('#password', 'Str0ng)Pass!')`, + `page.type('#password', 'type)value')`, + `page.locator('#pw').fill('abc)def')`, + `page.locator('#pw').fill('forced)value', { force: true })`, + `page.locator('#pw').pressSequentially(\`multi\nline)pass\`)`, + `page.keyboard.insertText("typed)secret")`, + `page.fill(buildSelector('nested)selector'), 'last)value')`, + ]; + const source = calls.join(";\n"); + const redacted = redactString(source); + + for (const secret of [ + "Str0ng)Pass!", + "type)value", + "abc)def", + "forced)value", + "multi\nline)pass", + "typed)secret", + "last)value", + ]) { + expect(redacted).not.toContain(secret); + expect(collectSensitiveValues({ code: source })).toContain(secret); + } + expect(redacted).toContain("buildSelector('nested)selector')"); + expect(redacted.match(/\[REDACTED\]/g)).toHaveLength(calls.length); + expect(() => assertSafeToPublish({ code: redacted })).not.toThrow(); + for (const call of calls) { + expect(() => assertSafeToPublish({ code: call })).toThrow( + "typed form value", + ); + } + }); }); describe("benchmark workflow hardening", () => { From 4767860a31310e3297a7237a1c6ac4f7811f3da8 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 30 Sep 2026 10:59:06 +0000 Subject: [PATCH 04/14] Cover additional benchmark secret fields --- benchmarks/harbor/redact.ts | 10 +++++----- benchmarks/harbor/results.test.ts | 28 ++++++++++++++++++++++++++++ 2 files changed, 33 insertions(+), 5 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 9ca20544..ce4050fc 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -16,7 +16,7 @@ const SENSITIVE_FIELDS = new Set([ "credential", "credentials", "cookie", - "set-cookie", + "set_cookie", "session_id", "replay_id", "cdp_url", @@ -36,7 +36,7 @@ function sensitiveField(key: string): boolean { const normalized = normalizedField(key); return ( SENSITIVE_FIELDS.has(normalized) || - /(?:^|_)(?:api_key|access_token|auth_token|refresh_token|session_token|password|private_key|credential|session_id|replay_id|cdp_url|viewer_url)$/.test( + /(?:^|_)(?:api_key|access_token|auth_token|refresh_token|session_token|password|private_key|credential|secret(?:_key)?|session_id|replay_id|cdp_url|viewer_url)$/.test( normalized, ) ); @@ -165,11 +165,11 @@ export function redactStringWithSecrets( .replace(/\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\b/g, REDACTED) .replace(/\b(?:sk|pk|bt|kapi|whsec)[-_][A-Za-z0-9_-]{12,}\b/gi, REDACTED) .replace( - /(["']?(?:api[_-]?key|access[_-]?token|auth[_-]?token|credential|jwt|password|private[_-]?key|refresh[_-]?token|replay[_-]?id|secret|session[_-]?id|session[_-]?token|token)["']?\s*[:=]\s*["']?)[^"'\s,}&]+/gi, + /(["']?(?:api[_-]?key|access[_-]?token|auth[_-]?token|credential|jwt|password|private[_-]?key|refresh[_-]?token|replay[_-]?id|[a-z0-9_-]*secret(?:[_-]?key)?|session[_-]?id|session[_-]?token|token)["']?\s*[:=]\s*["']?)[^"'\s,}&]+/gi, `$1${REDACTED}`, ) .replace( - /([?&](?:api[_-]?key|access[_-]?token|auth|code|credential|jwt|password|secret|session[_-]?id|session[_-]?token|token)=)[^&#\s]+/gi, + /([?&](?:api[_-]?key|access[_-]?token|auth|code|credential|jwt|password|[a-z0-9_-]*secret(?:[_-]?key)?|session[_-]?id|session[_-]?token|token)=)[^&#\s]+/gi, `$1${REDACTED}`, ) .replace(/(\b(?:cookie|set-cookie)\s*:\s*)[^\r\n]+/gi, `$1${REDACTED}`) @@ -358,7 +358,7 @@ function assertSafeString(value: string): void { } assertTypedCallsRedacted(value); for (const match of value.matchAll( - /["']?(?:api[_-]?key|access[_-]?token|auth[_-]?token|credential|jwt|password|private[_-]?key|refresh[_-]?token|replay[_-]?id|secret|session[_-]?id|session[_-]?token)["']?\s*[:=]\s*["']?([^"'\s,}&]+)/gi, + /["']?(?:api[_-]?key|access[_-]?token|auth[_-]?token|credential|jwt|password|private[_-]?key|refresh[_-]?token|replay[_-]?id|[a-z0-9_-]*secret(?:[_-]?key)?|session[_-]?id|session[_-]?token)["']?\s*[:=]\s*["']?([^"'\s,}&]+)/gi, )) { if (match[1] !== REDACTED) { throw new Error( diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index b7eb8b50..45388fa2 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -624,6 +624,34 @@ describe("Braintrust redaction", () => { delete process.env.TEST_API_KEY; }); + test("redacts normalized cookie headers and compound secret keys", () => { + const redacted = redactValue({ + "Set-Cookie": "session=visible", + client_secret: "client-value", + secret_key: "key-value", + webhook_secret: "webhook-value", + }); + expect(redacted).toEqual({ + "Set-Cookie": "[REDACTED]", + client_secret: "[REDACTED]", + secret_key: "[REDACTED]", + webhook_secret: "[REDACTED]", + }); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + expect(() => + assertSafeToPublish({ "Set-Cookie": "session=visible" }), + ).toThrow("Set-Cookie"); + expect(() => + assertSafeToPublish({ client_secret: "client-value" }), + ).toThrow("client_secret"); + + const text = redactString( + `'client_secret': 'client-value'&webhook_secret=webhook-value`, + ); + expect(text).not.toContain("client-value"); + expect(text).not.toContain("webhook-value"); + }); + test("redacts complex Playwright typing calls and rejects originals", () => { const calls = [ `page.fill('#password', 'Str0ng)Pass!')`, From ac21ec15498c6ae8b1f2a917331cbb08918cfc78 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 30 Sep 2026 15:43:01 +0000 Subject: [PATCH 05/14] Address benchmark trace follow-up --- benchmarks/harbor/README.md | 2 +- benchmarks/harbor/publish-braintrust.ts | 21 +-- benchmarks/harbor/redact.ts | 195 ++++++++++++------------ benchmarks/harbor/results.test.ts | 94 +++++++++++- 4 files changed, 203 insertions(+), 109 deletions(-) diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index 0e3d0bb5..08566136 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -59,7 +59,7 @@ Single-task runs have a 40-minute wall-clock limit. Full-suite runs default to s Before creating the Harbor dataset, the runner creates and deletes one disposable PurelyMail account. API errors stop the run before trials begin, so missing email accounts cannot silently become task failures. -The runner retries a whole isolated trial up to five times for transient Hypeman connection, timeout, exec-stream, and agent/task setup failures. Set `HARBOR_MAX_RETRIES` to override that limit. Per-request SDK retries remain disabled because transparently retrying instance or image creation can duplicate a request whose first response was lost. +The runner retries a whole isolated trial up to five times for transient Hypeman connection, timeout, exec-stream, and agent setup failures. Harbor step `setup.sh` failures are not retried; the PurelyMail preflight catches the known external setup dependency before trials begin. Set `HARBOR_MAX_RETRIES` to override the trial retry limit. Per-request SDK retries remain disabled because transparently retrying instance or image creation can duplicate a request whose first response was lost. ## GitHub Actions diff --git a/benchmarks/harbor/publish-braintrust.ts b/benchmarks/harbor/publish-braintrust.ts index cb42ac90..8461133a 100644 --- a/benchmarks/harbor/publish-braintrust.ts +++ b/benchmarks/harbor/publish-braintrust.ts @@ -234,10 +234,7 @@ function llmInput( break; } } - const context = - previousAgent === -1 - ? prior - : prior.slice(previousAgent, previousAgent + 1); + const context = previousAgent === -1 ? prior : prior.slice(previousAgent); return context.map((step) => stepContext(step, sensitiveValues)); } @@ -264,10 +261,14 @@ function privateInfoValues(steps: AtifStep[]): string[] { (result) => result.source_call_id === call.tool_call_id, ); if (typeof observation?.content !== "string") continue; - for (const match of observation.content.matchAll( - /["']\s*:\s*["']([^"'\\]{4,})["']/g, - )) { - values.add(match[1]); + let privateInfo: unknown = observation.content; + try { + privateInfo = JSON.parse(observation.content); + } catch { + // collectSensitiveValues also handles key/value pairs in non-JSON output. + } + for (const value of collectSensitiveValues(privateInfo)) { + values.add(value); } } } @@ -507,7 +508,9 @@ export function buildExperimentEvents( events.push(...atifEvents(trial, rowId)); } } - assertSafeToPublish(events); + for (const event of events) { + assertSafeToPublish(event, `event ${event.id}`); + } return events; } diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index ce4050fc..860737df 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -1,3 +1,5 @@ +import ts from "typescript"; + const SECRET_NAME = /(API_KEY|TOKEN|SECRET|PASSWORD|PRIVATE_KEY|CREDENTIAL)/i; const REDACTED = "[REDACTED]"; const REDACTED_PRIVATE_INFO = "[REDACTED_PRIVATE_INFO]"; @@ -26,6 +28,11 @@ const SENSITIVE_FIELDS = new Set([ "email", "phone", "address", + "account_number", + "card_number", + "routing_number", + "ssn", + "tax_id", ]); function normalizedField(key: string): string { @@ -42,6 +49,28 @@ function sensitiveField(key: string): boolean { ); } +const SENSITIVE_ASSIGNMENT = + /(["']?)([a-z0-9_-]+)\1(\s*[:=]\s*)(?:(["'])((?:\\.|(?!\4)[\s\S])*)\4|([^"'\s,}&]+))/gi; +const SENSITIVE_QUERY_VALUE = /([?&])([a-z0-9_-]+)=([^&#\s]+)/gi; +const SENSITIVE_QUERY_ONLY_FIELDS = new Set(["auth", "code"]); + +function redactSensitiveAssignments(value: string): string { + return value + .replace( + SENSITIVE_ASSIGNMENT, + (match, keyQuote, key, separator, valueQuote) => + sensitiveField(key) + ? `${keyQuote}${key}${keyQuote}${separator}${valueQuote ?? ""}${REDACTED}${valueQuote ?? ""}` + : match, + ) + .replace(SENSITIVE_QUERY_VALUE, (match, prefix, key) => + sensitiveField(key) || + SENSITIVE_QUERY_ONLY_FIELDS.has(normalizedField(key)) + ? `${prefix}${key}=${REDACTED}` + : match, + ); +} + function secretValues(): string[] { return Object.entries(process.env) .filter( @@ -69,22 +98,6 @@ function typedLiterals(value: string): TypedLiteral[] { let cursor = 0; while (cursor < value.length) { - const quote = value[cursor]; - if (quote === '"' || quote === "'" || quote === "`") { - cursor += 1; - while (cursor < value.length) { - if (value[cursor] === "\\") { - cursor += 2; - } else if (value[cursor] === quote) { - cursor += 1; - break; - } else { - cursor += 1; - } - } - continue; - } - const match = value.slice(cursor).match(callStart); if (!match || !TYPING_METHODS.has(match[1])) { cursor += 1; @@ -160,18 +173,10 @@ export function redactStringWithSecrets( if (secret.length >= 4) redacted = redacted.split(secret).join(REDACTED); } - redacted = redactTypedLiterals(redacted) + redacted = redactSensitiveAssignments(redactTypedLiterals(redacted)) .replace(/\bBearer\s+[A-Za-z0-9._~+/=-]+/gi, `Bearer ${REDACTED}`) .replace(/\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\b/g, REDACTED) .replace(/\b(?:sk|pk|bt|kapi|whsec)[-_][A-Za-z0-9_-]{12,}\b/gi, REDACTED) - .replace( - /(["']?(?:api[_-]?key|access[_-]?token|auth[_-]?token|credential|jwt|password|private[_-]?key|refresh[_-]?token|replay[_-]?id|[a-z0-9_-]*secret(?:[_-]?key)?|session[_-]?id|session[_-]?token|token)["']?\s*[:=]\s*["']?)[^"'\s,}&]+/gi, - `$1${REDACTED}`, - ) - .replace( - /([?&](?:api[_-]?key|access[_-]?token|auth|code|credential|jwt|password|[a-z0-9_-]*secret(?:[_-]?key)?|session[_-]?id|session[_-]?token|token)=)[^&#\s]+/gi, - `$1${REDACTED}`, - ) .replace(/(\b(?:cookie|set-cookie)\s*:\s*)[^\r\n]+/gi, `$1${REDACTED}`) .replace(/(\/browser\/live\/)[^/?#\s]+/gi, `$1${REDACTED}`) .replace(/(wss?:\/\/)[^/@\s]+@/gi, `$1${REDACTED}@`) @@ -219,10 +224,11 @@ export function redactValue(value: unknown, maxStringLength = 20_000): unknown { export function collectSensitiveValues(value: unknown): string[] { const values = new Set(); const collectString = (text: string) => { - for (const match of text.matchAll( - /["']?(?:password|private[_-]?key|secret|session[_-]?id|replay[_-]?id)["']?\s*[:=]\s*["']([^"'\s,}&]{4,})/gi, - )) { - values.add(match[1]); + for (const match of text.matchAll(new RegExp(SENSITIVE_ASSIGNMENT))) { + const value = match[5] ?? match[6]; + if (sensitiveField(match[2]) && value.length >= 4) { + values.add(value); + } } for (const typedValue of typedCallValues(text)) { if (typedValue.length >= 4) values.add(typedValue); @@ -283,72 +289,57 @@ export function privateInfoRead(toolName: string, input: unknown): boolean { } function assertTypedCallsRedacted(value: string): void { - const callStart = /^\.(?:fill|type|pressSequentially|insertText)\s*\(/; - let cursor = 0; - - while (cursor < value.length) { - const quote = value[cursor]; - if (quote === '"' || quote === "'" || quote === "`") { - cursor += 1; - while (cursor < value.length) { - if (value[cursor] === "\\") { - cursor += 2; - } else if (value[cursor] === quote) { - cursor += 1; - break; - } else { - cursor += 1; - } + for (const match of value.matchAll( + /\.(fill|type|pressSequentially|insertText)\s*\(/g, + )) { + const source = ts.createSourceFile( + "typed-call.ts", + `receiver${value.slice(match.index)}`, + ts.ScriptTarget.Latest, + true, + ts.ScriptKind.TS, + ); + let target: ts.CallExpression | undefined; + const findTarget = (node: ts.Node): void => { + if ( + !target && + ts.isCallExpression(node) && + ts.isPropertyAccessExpression(node.expression) && + node.expression.name.text === match[1] && + node.expression.expression.getText(source) === "receiver" + ) { + target = node; + return; } - continue; - } - - const match = value.slice(cursor).match(callStart); - if (!match) { - cursor += 1; - continue; + ts.forEachChild(node, findTarget); + }; + findTarget(source); + if (!target) { + throw new Error("Braintrust payload contains an unvalidated typing call"); } - let callCursor = cursor + match[0].length; - let depth = 1; - let lastLiteral: string | undefined; - while (callCursor < value.length && depth > 0) { - const callQuote = value[callCursor]; - if (callQuote === '"' || callQuote === "'" || callQuote === "`") { - const literalStart = callCursor + 1; - callCursor += 1; - while (callCursor < value.length) { - if (value[callCursor] === "\\") { - callCursor += 2; - } else if (value[callCursor] === callQuote) { - break; - } else { - callCursor += 1; - } - } - lastLiteral = value.slice(literalStart, callCursor); - if (callCursor < value.length) callCursor += 1; - continue; - } - if (value.startsWith("//", callCursor)) { - const newline = value.indexOf("\n", callCursor + 2); - callCursor = newline === -1 ? value.length : newline + 1; - continue; + const literals: ts.Node[] = []; + const collectLiterals = (node: ts.Node): void => { + if ( + ts.isStringLiteral(node) || + ts.isNoSubstitutionTemplateLiteral(node) || + ts.isTemplateExpression(node) + ) { + literals.push(node); + return; } - if (value.startsWith("/*", callCursor)) { - const commentEnd = value.indexOf("*/", callCursor + 2); - callCursor = commentEnd === -1 ? value.length : commentEnd + 2; - continue; - } - if (value[callCursor] === "(") depth += 1; - if (value[callCursor] === ")") depth -= 1; - callCursor += 1; - } - - if (lastLiteral !== undefined && lastLiteral !== REDACTED) { + ts.forEachChild(node, collectLiterals); + }; + for (const argument of target.arguments) collectLiterals(argument); + const literal = literals.at(-1); + if (!literal) continue; + const literalValue = + ts.isStringLiteral(literal) || ts.isNoSubstitutionTemplateLiteral(literal) + ? literal.text + : literal.getText(source).slice(1, -1); + if (literalValue !== REDACTED) { throw new Error("Braintrust payload still contains a typed form value"); } - cursor = callCursor; } } @@ -357,10 +348,8 @@ function assertSafeString(value: string): void { throw new Error("Braintrust payload still contains an email address"); } assertTypedCallsRedacted(value); - for (const match of value.matchAll( - /["']?(?:api[_-]?key|access[_-]?token|auth[_-]?token|credential|jwt|password|private[_-]?key|refresh[_-]?token|replay[_-]?id|[a-z0-9_-]*secret(?:[_-]?key)?|session[_-]?id|session[_-]?token)["']?\s*[:=]\s*["']?([^"'\s,}&]+)/gi, - )) { - if (match[1] !== REDACTED) { + for (const match of value.matchAll(new RegExp(SENSITIVE_ASSIGNMENT))) { + if (sensitiveField(match[2]) && (match[5] ?? match[6]) !== REDACTED) { throw new Error( "Braintrust payload still contains a sensitive field value", ); @@ -373,23 +362,33 @@ function assertSafeString(value: string): void { } } -export function assertSafeToPublish(value: unknown): void { +export function assertSafeToPublish(value: unknown, path = "$"): void { if (typeof value === "string") { - assertSafeString(value); + try { + assertSafeString(value); + } catch (error) { + const message = error instanceof Error ? error.message : String(error); + throw new Error(`${message} at ${path}`); + } return; } if (Array.isArray(value)) { - for (const entry of value) assertSafeToPublish(entry); + for (const [index, entry] of value.entries()) { + assertSafeToPublish(entry, `${path}[${index}]`); + } return; } if (value !== null && typeof value === "object") { for (const [key, entry] of Object.entries( value as Record, )) { + const childPath = /^[A-Za-z_$][A-Za-z0-9_$]*$/.test(key) + ? `${path}.${key}` + : `${path}[${JSON.stringify(key)}]`; if (sensitiveField(key) && entry !== REDACTED) { - throw new Error(`Braintrust payload did not redact ${key}`); + throw new Error(`Braintrust payload did not redact ${childPath}`); } - assertSafeToPublish(entry); + assertSafeToPublish(entry, childPath); } } } diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 45388fa2..3ee79886 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -566,6 +566,86 @@ describe("Harbor result ingestion", () => { ).filter((event) => event.span_attributes.type === "llm"); expect(new Set(events.map((event) => event.id)).size).toBe(2); }); + + test("includes context added after the previous agent turn", () => { + const root = fixture(); + writeJson(join(root, "task-one__abc", "steps/run/agent/trajectory.json"), { + steps: [ + { source: "system", message: "system prompt" }, + { source: "user", message: "first request" }, + { source: "agent", message: "first response" }, + { source: "user", message: "follow-up" }, + { source: "system", message: "updated constraint" }, + { source: "agent", message: "second response" }, + ], + }); + const llmEvents = buildExperimentEvents( + [readBenchmarkArm({ name: "candidate", path: root })], + "multi-turn-context", + ).filter((event) => event.span_attributes.type === "llm"); + + expect(llmEvents[1].input).toEqual([ + { + source: "agent", + message: "first response", + toolCalls: [], + observations: [], + }, + { + source: "user", + message: "follow-up", + toolCalls: [], + observations: [], + }, + { + source: "system", + message: "updated constraint", + toolCalls: [], + observations: [], + }, + ]); + }); + + test("harvests only sensitive values from private-info output", () => { + const root = fixture(); + writeJson(join(root, "task-one__abc", "steps/run/agent/trajectory.json"), { + steps: [ + { source: "user", message: "perform the task" }, + { + source: "agent", + message: "", + tool_calls: [ + { + tool_call_id: "private-read", + function_name: "Read", + arguments: { file_path: "/my-info/personal.json" }, + }, + ], + observation: { + results: [ + { + source_call_id: "private-read", + content: + '{"name":"Close Window","width":"1208","account_number":"12345678"}', + }, + ], + }, + }, + { + source: "agent", + message: "Close Window width 1208 account 12345678", + }, + ], + }); + const llmEvents = buildExperimentEvents( + [readBenchmarkArm({ name: "candidate", path: root })], + "private-info-values", + ).filter((event) => event.span_attributes.type === "llm"); + + expect(llmEvents[1].output).toBe( + "Close Window width 1208 account [REDACTED]", + ); + }); }); describe("Braintrust redaction", () => { @@ -601,6 +681,11 @@ describe("Braintrust redaction", () => { code: "page.fill('#password', 'still-visible')", }), ).toThrow("typed form value"); + expect(() => + assertSafeToPublish({ + nested: { code: "page.fill('#password', 'still-visible')" }, + }), + ).toThrow("$.nested.code"); expect( privateInfoRead("exec_command", { cmd: "cat /my-info/email_credentials.json", @@ -646,10 +731,11 @@ describe("Braintrust redaction", () => { ).toThrow("client_secret"); const text = redactString( - `'client_secret': 'client-value'&webhook_secret=webhook-value`, + `'client_secret': 'client-value'&webhook_secret=webhook-value; 'address': '123 Main St'`, ); expect(text).not.toContain("client-value"); expect(text).not.toContain("webhook-value"); + expect(text).not.toContain("123 Main St"); }); test("redacts complex Playwright typing calls and rejects originals", () => { @@ -661,6 +747,9 @@ describe("Braintrust redaction", () => { `page.locator('#pw').pressSequentially(\`multi\nline)pass\`)`, `page.keyboard.insertText("typed)secret")`, `page.fill(buildSelector('nested)selector'), 'last)value')`, + `// Fill in the user's password\nawait page.locator('#password').fill('Comment2Pass')`, + `/* we'll sign up now */\nawait page.fill('#password', 'Block2Pass')`, + `node -e 'await page.fill("#pw", "Shell2Pass")'`, ]; const source = calls.join(";\n"); const redacted = redactString(source); @@ -673,6 +762,9 @@ describe("Braintrust redaction", () => { "multi\nline)pass", "typed)secret", "last)value", + "Comment2Pass", + "Block2Pass", + "Shell2Pass", ]) { expect(redacted).not.toContain(secret); expect(collectSensitiveValues({ code: source })).toContain(secret); From 8975eb9c45485bdb5124f4ca4ceddd377e93f6ec Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 30 Sep 2026 15:54:59 +0000 Subject: [PATCH 06/14] Normalize camelCase benchmark secret keys --- benchmarks/harbor/redact.ts | 6 +++++- benchmarks/harbor/results.test.ts | 26 ++++++++++++++++++++++---- 2 files changed, 27 insertions(+), 5 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 860737df..5139af35 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -36,7 +36,11 @@ const SENSITIVE_FIELDS = new Set([ ]); function normalizedField(key: string): string { - return key.trim().toLowerCase().replace(/-/g, "_"); + return key + .trim() + .replace(/([a-z0-9])([A-Z])/g, "$1_$2") + .toLowerCase() + .replace(/-/g, "_"); } function sensitiveField(key: string): boolean { diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 3ee79886..4396cba3 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -715,12 +715,20 @@ describe("Braintrust redaction", () => { client_secret: "client-value", secret_key: "key-value", webhook_secret: "webhook-value", + apiKey: "camel-api-value", + sessionId: "camel-session-value", + clientSecret: "camel-secret-value", + accountNumber: "camel-account-value", }); expect(redacted).toEqual({ "Set-Cookie": "[REDACTED]", client_secret: "[REDACTED]", secret_key: "[REDACTED]", webhook_secret: "[REDACTED]", + apiKey: "[REDACTED]", + sessionId: "[REDACTED]", + clientSecret: "[REDACTED]", + accountNumber: "[REDACTED]", }); expect(() => assertSafeToPublish(redacted)).not.toThrow(); expect(() => @@ -729,13 +737,23 @@ describe("Braintrust redaction", () => { expect(() => assertSafeToPublish({ client_secret: "client-value" }), ).toThrow("client_secret"); + expect(() => + assertSafeToPublish({ clientSecret: "camel-secret-value" }), + ).toThrow("clientSecret"); const text = redactString( - `'client_secret': 'client-value'&webhook_secret=webhook-value; 'address': '123 Main St'`, + `'client_secret': 'client-value'&webhook_secret=webhook-value; 'address': '123 Main St'; "apiKey":"camel-api-value"; "sessionId":"camel-session-value"; "clientSecret":"camel-secret-value"`, ); - expect(text).not.toContain("client-value"); - expect(text).not.toContain("webhook-value"); - expect(text).not.toContain("123 Main St"); + for (const secret of [ + "client-value", + "webhook-value", + "123 Main St", + "camel-api-value", + "camel-session-value", + "camel-secret-value", + ]) { + expect(text).not.toContain(secret); + } }); test("redacts complex Playwright typing calls and rejects originals", () => { From 3538a03c9d365b1717e6149e663af59548f99068 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 30 Sep 2026 17:11:22 +0000 Subject: [PATCH 07/14] Harden nested benchmark redaction --- benchmarks/harbor/publish-braintrust.ts | 5 +- benchmarks/harbor/redact.ts | 61 +++++++++++++++++++++++-- benchmarks/harbor/results.test.ts | 30 ++++++++++-- 3 files changed, 86 insertions(+), 10 deletions(-) diff --git a/benchmarks/harbor/publish-braintrust.ts b/benchmarks/harbor/publish-braintrust.ts index 8461133a..2f7d7347 100644 --- a/benchmarks/harbor/publish-braintrust.ts +++ b/benchmarks/harbor/publish-braintrust.ts @@ -13,6 +13,7 @@ import { } from "./results"; import { assertSafeToPublish, + collectPrivateInfoValues, collectSensitiveValues, privateInfoRead, redactString, @@ -265,9 +266,9 @@ function privateInfoValues(steps: AtifStep[]): string[] { try { privateInfo = JSON.parse(observation.content); } catch { - // collectSensitiveValues also handles key/value pairs in non-JSON output. + // The collector also handles key/value pairs in non-JSON output. } - for (const value of collectSensitiveValues(privateInfo)) { + for (const value of collectPrivateInfoValues(privateInfo)) { values.add(value); } } diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 5139af35..ee6a7871 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -54,7 +54,7 @@ function sensitiveField(key: string): boolean { } const SENSITIVE_ASSIGNMENT = - /(["']?)([a-z0-9_-]+)\1(\s*[:=]\s*)(?:(["'])((?:\\.|(?!\4)[\s\S])*)\4|([^"'\s,}&]+))/gi; + /(["']?)([a-z0-9_-]+)\1(\s*[:=]\s*)(?:(["'])((?:\\[\s\S]|(?!\4)[^\\])*)\4|([^"'\s,}&]+))/gi; const SENSITIVE_QUERY_VALUE = /([?&])([a-z0-9_-]+)=([^&#\s]+)/gi; const SENSITIVE_QUERY_ONLY_FIELDS = new Set(["auth", "code"]); @@ -62,10 +62,15 @@ function redactSensitiveAssignments(value: string): string { return value .replace( SENSITIVE_ASSIGNMENT, - (match, keyQuote, key, separator, valueQuote) => - sensitiveField(key) - ? `${keyQuote}${key}${keyQuote}${separator}${valueQuote ?? ""}${REDACTED}${valueQuote ?? ""}` - : match, + (match, keyQuote, key, separator, valueQuote, quotedValue) => { + if (sensitiveField(key)) { + return `${keyQuote}${key}${keyQuote}${separator}${valueQuote ?? ""}${REDACTED}${valueQuote ?? ""}`; + } + if (valueQuote) { + return `${keyQuote}${key}${keyQuote}${separator}${valueQuote}${redactSensitiveAssignments(quotedValue)}${valueQuote}`; + } + return match; + }, ) .replace(SENSITIVE_QUERY_VALUE, (match, prefix, key) => sensitiveField(key) || @@ -232,6 +237,8 @@ export function collectSensitiveValues(value: unknown): string[] { const value = match[5] ?? match[6]; if (sensitiveField(match[2]) && value.length >= 4) { values.add(value); + } else if (match[4]) { + collectString(value); } } for (const typedValue of typedCallValues(text)) { @@ -265,6 +272,47 @@ export function collectSensitiveValues(value: unknown): string[] { return [...values].sort((left, right) => right.length - left.length); } +const PRIVATE_INFO_CONTEXTS = new Set([ + "financial", + "government_ids", + "insurance", +]); +const PRIVATE_INFO_IDENTIFIERS = new Set([ + "number", + "number_formatted", + "sin", + "transit_number", +]); + +export function collectPrivateInfoValues(value: unknown): string[] { + const values = new Set(collectSensitiveValues(value)); + const visit = (entry: unknown, inSensitiveContext = false): void => { + if (Array.isArray(entry)) { + for (const item of entry) visit(item, inSensitiveContext); + return; + } + if (entry === null || typeof entry !== "object") return; + for (const [key, child] of Object.entries( + entry as Record, + )) { + const normalized = normalizedField(key); + const childContext = + inSensitiveContext || PRIVATE_INFO_CONTEXTS.has(normalized); + if ( + childContext && + PRIVATE_INFO_IDENTIFIERS.has(normalized) && + typeof child === "string" && + child.length >= 4 + ) { + values.add(child); + } + visit(child, childContext); + } + }; + visit(value); + return [...values].sort((left, right) => right.length - left.length); +} + export function privateInfoRead(toolName: string, input: unknown): boolean { if (!/(?:^|__)(?:exec_command|bash|read)$/i.test(toolName)) return false; const fields = @@ -358,6 +406,9 @@ function assertSafeString(value: string): void { "Braintrust payload still contains a sensitive field value", ); } + if (!sensitiveField(match[2]) && match[4]) { + assertSafeString(match[5]); + } } for (const secret of secretValues()) { if (value.includes(secret)) { diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 4396cba3..32866594 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -626,14 +626,15 @@ describe("Harbor result ingestion", () => { { source_call_id: "private-read", content: - '{"name":"Close Window","width":"1208","account_number":"12345678"}', + '{"name":"Close Window","width":"1208","account_number":"12345678","government_ids":{"passport":{"number":"JK456789"},"drivers_license":{"number":"G4567-89018-05501"},"health_card":{"number":"6789-012-345"},"sin":"472-345-678"},"financial":{"bank_accounts":[{"transit_number":"10202"}],"credit_cards":[{"number":"4519873424604532"}]}}', }, ], }, }, { source: "agent", - message: "Close Window width 1208 account 12345678", + message: + "Close Window width 1208 account 12345678 passport JK456789 licence G4567-89018-05501 health 6789-012-345 sin 472-345-678 transit 10202 card 4519873424604532", }, ], }); @@ -643,7 +644,7 @@ describe("Harbor result ingestion", () => { ).filter((event) => event.span_attributes.type === "llm"); expect(llmEvents[1].output).toBe( - "Close Window width 1208 account [REDACTED]", + "Close Window width 1208 account [REDACTED] passport [REDACTED] licence [REDACTED] health [REDACTED] sin [REDACTED] transit [REDACTED] card [REDACTED]", ); }); }); @@ -756,6 +757,29 @@ describe("Braintrust redaction", () => { } }); + test("redacts nested and unterminated sensitive assignments", () => { + const nested = + '{"success": true, "result": "Account created.\\nTemporary password: Hunter2Pass"}'; + const unterminated = + '{"text": "Step 1...' + + "\\n".repeat(30) + + "...password: Unterminated2Pass"; + + for (const [source, secret] of [ + [nested, "Hunter2Pass"], + [unterminated, "Unterminated2Pass"], + ]) { + const redacted = redactString(source); + expect(redacted).not.toContain(secret); + expect(redacted).toContain("password: [REDACTED]"); + expect(collectSensitiveValues(source)).toContain(secret); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + expect(() => assertSafeToPublish(source)).toThrow( + "sensitive field value", + ); + } + }); + test("redacts complex Playwright typing calls and rejects originals", () => { const calls = [ `page.fill('#password', 'Str0ng)Pass!')`, From 34f9e1e237ba1f5b9994a0d079cf7c6c6cce5535 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 30 Sep 2026 17:33:20 +0000 Subject: [PATCH 08/14] Redact escaped nested benchmark secrets --- benchmarks/harbor/redact.ts | 61 ++++++++++++++++++++++++------- benchmarks/harbor/results.test.ts | 7 +++- 2 files changed, 54 insertions(+), 14 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index ee6a7871..5d6fe846 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -54,22 +54,54 @@ function sensitiveField(key: string): boolean { } const SENSITIVE_ASSIGNMENT = - /(["']?)([a-z0-9_-]+)\1(\s*[:=]\s*)(?:(["'])((?:\\[\s\S]|(?!\4)[^\\])*)\4|([^"'\s,}&]+))/gi; + /(["']?)([a-z0-9_-]+)\1(\s*[:=]\s*)(?:(["'])((?:\\[\s\S]|(?!\4)[^\\])*)\4|\\(["'])((?:\\(?!\6)[\s\S]|[^\\])*)\\\6|([^"'\s,}&]+))/gi; const SENSITIVE_QUERY_VALUE = /([?&])([a-z0-9_-]+)=([^&#\s]+)/gi; const SENSITIVE_QUERY_ONLY_FIELDS = new Set(["auth", "code"]); +function decodeQuotedValue(value: string): { + decoded: string; + encode: (decoded: string) => string; +} { + const unchanged = { decoded: value, encode: (updated: string) => updated }; + try { + const decoded: unknown = JSON.parse(`"${value}"`); + return typeof decoded === "string" + ? { + decoded, + encode: (updated) => JSON.stringify(updated).slice(1, -1), + } + : unchanged; + } catch { + return unchanged; + } +} + function redactSensitiveAssignments(value: string): string { return value .replace( SENSITIVE_ASSIGNMENT, - (match, keyQuote, key, separator, valueQuote, quotedValue) => { + ( + match, + keyQuote, + key, + separator, + valueQuote, + quotedValue, + escapedValueQuote, + escapedQuotedValue, + ) => { + const quote = valueQuote ?? escapedValueQuote; + const quoted = quotedValue ?? escapedQuotedValue; + const delimiter = escapedValueQuote ? `\\${quote}` : (quote ?? ""); if (sensitiveField(key)) { - return `${keyQuote}${key}${keyQuote}${separator}${valueQuote ?? ""}${REDACTED}${valueQuote ?? ""}`; - } - if (valueQuote) { - return `${keyQuote}${key}${keyQuote}${separator}${valueQuote}${redactSensitiveAssignments(quotedValue)}${valueQuote}`; + return `${keyQuote}${key}${keyQuote}${separator}${delimiter}${REDACTED}${delimiter}`; } - return match; + if (!quote) return match; + const { decoded, encode } = decodeQuotedValue(quoted); + const redacted = redactSensitiveAssignments(decoded); + return redacted === decoded + ? match + : `${keyQuote}${key}${keyQuote}${separator}${delimiter}${encode(redacted)}${delimiter}`; }, ) .replace(SENSITIVE_QUERY_VALUE, (match, prefix, key) => @@ -234,11 +266,11 @@ export function collectSensitiveValues(value: unknown): string[] { const values = new Set(); const collectString = (text: string) => { for (const match of text.matchAll(new RegExp(SENSITIVE_ASSIGNMENT))) { - const value = match[5] ?? match[6]; + const value = match[5] ?? match[7] ?? match[8]; if (sensitiveField(match[2]) && value.length >= 4) { values.add(value); - } else if (match[4]) { - collectString(value); + } else if (match[4] || match[6]) { + collectString(decodeQuotedValue(value).decoded); } } for (const typedValue of typedCallValues(text)) { @@ -401,13 +433,16 @@ function assertSafeString(value: string): void { } assertTypedCallsRedacted(value); for (const match of value.matchAll(new RegExp(SENSITIVE_ASSIGNMENT))) { - if (sensitiveField(match[2]) && (match[5] ?? match[6]) !== REDACTED) { + if ( + sensitiveField(match[2]) && + (match[5] ?? match[7] ?? match[8]) !== REDACTED + ) { throw new Error( "Braintrust payload still contains a sensitive field value", ); } - if (!sensitiveField(match[2]) && match[4]) { - assertSafeString(match[5]); + if (!sensitiveField(match[2]) && (match[4] || match[6])) { + assertSafeString(decodeQuotedValue(match[5] ?? match[7]).decoded); } } for (const secret of secretValues()) { diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 32866594..0a0ff74e 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -764,14 +764,19 @@ describe("Braintrust redaction", () => { '{"text": "Step 1...' + "\\n".repeat(30) + "...password: Unterminated2Pass"; + const escapedNested = + '{"success":true,"result":"{\\"password\\":\\"Escaped2Pass\\"}"}'; + const escapedValue = String.raw`password: \"EscapedValue2Pass\"`; for (const [source, secret] of [ [nested, "Hunter2Pass"], [unterminated, "Unterminated2Pass"], + [escapedNested, "Escaped2Pass"], + [escapedValue, "EscapedValue2Pass"], ]) { const redacted = redactString(source); expect(redacted).not.toContain(secret); - expect(redacted).toContain("password: [REDACTED]"); + expect(redacted).toContain("[REDACTED]"); expect(collectSensitiveValues(source)).toContain(secret); expect(() => assertSafeToPublish(redacted)).not.toThrow(); expect(() => assertSafeToPublish(source)).toThrow( From 49406de6d6326b55366ad06858e8e06a8721ad9c Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:00:19 +0000 Subject: [PATCH 09/14] Scan nested benchmark assignments safely --- benchmarks/harbor/redact.ts | 167 ++++++++++++++++++------------ benchmarks/harbor/results.test.ts | 9 +- 2 files changed, 105 insertions(+), 71 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 5d6fe846..e0803373 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -53,63 +53,88 @@ function sensitiveField(key: string): boolean { ); } -const SENSITIVE_ASSIGNMENT = - /(["']?)([a-z0-9_-]+)\1(\s*[:=]\s*)(?:(["'])((?:\\[\s\S]|(?!\4)[^\\])*)\4|\\(["'])((?:\\(?!\6)[\s\S]|[^\\])*)\\\6|([^"'\s,}&]+))/gi; const SENSITIVE_QUERY_VALUE = /([?&])([a-z0-9_-]+)=([^&#\s]+)/gi; const SENSITIVE_QUERY_ONLY_FIELDS = new Set(["auth", "code"]); -function decodeQuotedValue(value: string): { - decoded: string; - encode: (decoded: string) => string; -} { - const unchanged = { decoded: value, encode: (updated: string) => updated }; - try { - const decoded: unknown = JSON.parse(`"${value}"`); - return typeof decoded === "string" - ? { - decoded, - encode: (updated) => JSON.stringify(updated).slice(1, -1), +interface SensitiveAssignment { + start: number; + end: number; + value: string; +} + +function sensitiveAssignments(text: string): SensitiveAssignment[] { + const assignments: SensitiveAssignment[] = []; + for (let separator = 0; separator < text.length; separator += 1) { + if (text[separator] !== ":" && text[separator] !== "=") continue; + + let keyCursor = separator - 1; + while (/\s/.test(text[keyCursor] ?? "")) keyCursor -= 1; + if (text[keyCursor] === '"' || text[keyCursor] === "'") { + keyCursor -= 1; + while (text[keyCursor] === "\\") keyCursor -= 1; + } + const keyEnd = keyCursor + 1; + while (/[A-Za-z0-9_-]/.test(text[keyCursor] ?? "")) { + if (/[nrt]/.test(text[keyCursor] ?? "") && text[keyCursor - 1] === "\\") { + break; + } + keyCursor -= 1; + } + const key = text.slice(keyCursor + 1, keyEnd); + if (!key || !sensitiveField(key)) continue; + + let valueCursor = separator + 1; + while (/\s/.test(text[valueCursor] ?? "")) valueCursor += 1; + let slashCount = 0; + while (text[valueCursor + slashCount] === "\\") slashCount += 1; + const quote = text[valueCursor + slashCount]; + let start = valueCursor; + let end = valueCursor; + + if (quote === '"' || quote === "'") { + start = valueCursor + slashCount + 1; + end = text.length; + for (let cursor = start; cursor < text.length; cursor += 1) { + if (text[cursor] !== quote) continue; + let precedingSlashes = 0; + for ( + let slashCursor = cursor - 1; + text[slashCursor] === "\\"; + slashCursor -= 1 + ) { + precedingSlashes += 1; } - : unchanged; - } catch { - return unchanged; + if (precedingSlashes === slashCount) { + end = cursor - slashCount; + separator = cursor; + break; + } + } + if (end === text.length) separator = text.length; + } else { + while (end < text.length && !/[\s,"'\}&;]/.test(text[end] ?? "")) { + end += 1; + } + separator = Math.max(separator, end - 1); + } + + if (end > start) { + assignments.push({ start, end, value: text.slice(start, end) }); + } } + return assignments; } function redactSensitiveAssignments(value: string): string { - return value - .replace( - SENSITIVE_ASSIGNMENT, - ( - match, - keyQuote, - key, - separator, - valueQuote, - quotedValue, - escapedValueQuote, - escapedQuotedValue, - ) => { - const quote = valueQuote ?? escapedValueQuote; - const quoted = quotedValue ?? escapedQuotedValue; - const delimiter = escapedValueQuote ? `\\${quote}` : (quote ?? ""); - if (sensitiveField(key)) { - return `${keyQuote}${key}${keyQuote}${separator}${delimiter}${REDACTED}${delimiter}`; - } - if (!quote) return match; - const { decoded, encode } = decodeQuotedValue(quoted); - const redacted = redactSensitiveAssignments(decoded); - return redacted === decoded - ? match - : `${keyQuote}${key}${keyQuote}${separator}${delimiter}${encode(redacted)}${delimiter}`; - }, - ) - .replace(SENSITIVE_QUERY_VALUE, (match, prefix, key) => - sensitiveField(key) || - SENSITIVE_QUERY_ONLY_FIELDS.has(normalizedField(key)) - ? `${prefix}${key}=${REDACTED}` - : match, - ); + let redacted = value; + for (const assignment of sensitiveAssignments(value).reverse()) { + redacted = `${redacted.slice(0, assignment.start)}${REDACTED}${redacted.slice(assignment.end)}`; + } + return redacted.replace(SENSITIVE_QUERY_VALUE, (match, prefix, key) => + sensitiveField(key) || SENSITIVE_QUERY_ONLY_FIELDS.has(normalizedField(key)) + ? `${prefix}${key}=${REDACTED}` + : match, + ); } function secretValues(): string[] { @@ -265,13 +290,8 @@ export function redactValue(value: unknown, maxStringLength = 20_000): unknown { export function collectSensitiveValues(value: unknown): string[] { const values = new Set(); const collectString = (text: string) => { - for (const match of text.matchAll(new RegExp(SENSITIVE_ASSIGNMENT))) { - const value = match[5] ?? match[7] ?? match[8]; - if (sensitiveField(match[2]) && value.length >= 4) { - values.add(value); - } else if (match[4] || match[6]) { - collectString(decodeQuotedValue(value).decoded); - } + for (const assignment of sensitiveAssignments(text)) { + if (assignment.value.length >= 4) values.add(assignment.value); } for (const typedValue of typedCallValues(text)) { if (typedValue.length >= 4) values.add(typedValue); @@ -427,24 +447,35 @@ function assertTypedCallsRedacted(value: string): void { } } -function assertSafeString(value: string): void { - if (/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/i.test(value)) { - throw new Error("Braintrust payload still contains an email address"); - } - assertTypedCallsRedacted(value); - for (const match of value.matchAll(new RegExp(SENSITIVE_ASSIGNMENT))) { - if ( - sensitiveField(match[2]) && - (match[5] ?? match[7] ?? match[8]) !== REDACTED - ) { +function assertSensitiveAssignmentsRedacted(value: string): void { + const assignmentStart = + /(?:^|\\[nrt]|[^A-Za-z0-9_-])(?:\\*["'])?([A-Za-z0-9_-]+)(?:\\*["'])?\s*[:=]\s*/gi; + for (const match of value.matchAll(assignmentStart)) { + if (!sensitiveField(match[1])) continue; + const remainder = value.slice((match.index ?? 0) + match[0].length); + const unquoted = remainder.replace(/^(?:\\*["'])?/, ""); + if (!unquoted.startsWith(REDACTED)) { throw new Error( "Braintrust payload still contains a sensitive field value", ); } - if (!sensitiveField(match[2]) && (match[4] || match[6])) { - assertSafeString(decodeQuotedValue(match[5] ?? match[7]).decoded); + const afterRedaction = unquoted + .slice(REDACTED.length) + .replace(/^\\*["']/, ""); + if (/^[A-Za-z0-9]/.test(afterRedaction)) { + throw new Error( + "Braintrust payload still contains data after a redacted field value", + ); } } +} + +function assertSafeString(value: string): void { + if (/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/i.test(value)) { + throw new Error("Braintrust payload still contains an email address"); + } + assertTypedCallsRedacted(value); + assertSensitiveAssignmentsRedacted(value); for (const secret of secretValues()) { if (value.includes(secret)) { throw new Error("Braintrust payload still contains a configured secret"); diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 0a0ff74e..0a5fc65d 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -761,18 +761,18 @@ describe("Braintrust redaction", () => { const nested = '{"success": true, "result": "Account created.\\nTemporary password: Hunter2Pass"}'; const unterminated = - '{"text": "Step 1...' + - "\\n".repeat(30) + - "...password: Unterminated2Pass"; + '{"text": "Step 1...' + "\\n".repeat(30) + "password: Unterminated2Pass"; const escapedNested = '{"success":true,"result":"{\\"password\\":\\"Escaped2Pass\\"}"}'; const escapedValue = String.raw`password: \"EscapedValue2Pass\"`; + const escapedWrapped = String.raw`result: \"wrapper password: \\\"EscapedWrapped2Pass\\\"\"`; for (const [source, secret] of [ [nested, "Hunter2Pass"], [unterminated, "Unterminated2Pass"], [escapedNested, "Escaped2Pass"], [escapedValue, "EscapedValue2Pass"], + [escapedWrapped, "EscapedWrapped2Pass"], ]) { const redacted = redactString(source); expect(redacted).not.toContain(secret); @@ -783,6 +783,9 @@ describe("Braintrust redaction", () => { "sensitive field value", ); } + expect(() => + assertSafeToPublish('password: [REDACTED]"StillVisible"'), + ).toThrow("data after a redacted field value"); }); test("redacts complex Playwright typing calls and rejects originals", () => { From d9ef3455416b9fb3ceae56d7db15bd013efd3a95 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:13:34 +0000 Subject: [PATCH 10/14] Handle empty and structured secret values --- benchmarks/harbor/redact.ts | 33 ++++++++++++++++++++++++++++--- benchmarks/harbor/results.test.ts | 14 +++++++++++++ 2 files changed, 44 insertions(+), 3 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index e0803373..6043948c 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -62,6 +62,31 @@ interface SensitiveAssignment { value: string; } +function structuredValueEnd(text: string, start: number): number { + const stack = [text[start]]; + let quote: string | undefined; + for (let cursor = start + 1; cursor < text.length; cursor += 1) { + const character = text[cursor]; + if (quote) { + if (character === "\\") cursor += 1; + else if (character === quote) quote = undefined; + continue; + } + if (character === '"' || character === "'") { + quote = character; + } else if (character === "{" || character === "[") { + stack.push(character); + } else if ( + (character === "}" && stack.at(-1) === "{") || + (character === "]" && stack.at(-1) === "[") + ) { + stack.pop(); + if (stack.length === 0) return cursor + 1; + } + } + return text.length; +} + function sensitiveAssignments(text: string): SensitiveAssignment[] { const assignments: SensitiveAssignment[] = []; for (let separator = 0; separator < text.length; separator += 1) { @@ -111,6 +136,10 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { } } if (end === text.length) separator = text.length; + } else if (quote === "{" || quote === "[") { + start = valueCursor; + end = structuredValueEnd(text, start); + separator = end - 1; } else { while (end < text.length && !/[\s,"'\}&;]/.test(text[end] ?? "")) { end += 1; @@ -118,9 +147,7 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { separator = Math.max(separator, end - 1); } - if (end > start) { - assignments.push({ start, end, value: text.slice(start, end) }); - } + assignments.push({ start, end, value: text.slice(start, end) }); } return assignments; } diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 0a5fc65d..f51f1b77 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -786,6 +786,20 @@ describe("Braintrust redaction", () => { expect(() => assertSafeToPublish('password: [REDACTED]"StillVisible"'), ).toThrow("data after a redacted field value"); + + for (const source of [ + 'password: ""', + "token=", + 'credentials: {"username":"alex","password":"Nested2Pass"}', + 'credentials: ["first", {"token":"NestedToken"}]', + ]) { + const redacted = redactString(source); + expect(redacted).toContain("[REDACTED]"); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + expect(() => assertSafeToPublish(source)).toThrow( + "sensitive field value", + ); + } }); test("redacts complex Playwright typing calls and rejects originals", () => { From 99d98e378a9963a10935180e6b8a311363c1f28c Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:24:06 +0000 Subject: [PATCH 11/14] Collect secrets nested in structured values --- benchmarks/harbor/redact.ts | 1 + benchmarks/harbor/results.test.ts | 12 ++++++------ 2 files changed, 7 insertions(+), 6 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 6043948c..479a5a32 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -319,6 +319,7 @@ export function collectSensitiveValues(value: unknown): string[] { const collectString = (text: string) => { for (const assignment of sensitiveAssignments(text)) { if (assignment.value.length >= 4) values.add(assignment.value); + if (/^[{[]/.test(assignment.value)) collectString(assignment.value); } for (const typedValue of typedCallValues(text)) { if (typedValue.length >= 4) values.add(typedValue); diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index f51f1b77..f5cba6ba 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -787,12 +787,10 @@ describe("Braintrust redaction", () => { assertSafeToPublish('password: [REDACTED]"StillVisible"'), ).toThrow("data after a redacted field value"); - for (const source of [ - 'password: ""', - "token=", - 'credentials: {"username":"alex","password":"Nested2Pass"}', - 'credentials: ["first", {"token":"NestedToken"}]', - ]) { + const objectValue = + 'credentials: {"username":"alex","password":"Nested2Pass"}'; + const arrayValue = 'credentials: ["first", {"token":"NestedToken"}]'; + for (const source of ['password: ""', "token=", objectValue, arrayValue]) { const redacted = redactString(source); expect(redacted).toContain("[REDACTED]"); expect(() => assertSafeToPublish(redacted)).not.toThrow(); @@ -800,6 +798,8 @@ describe("Braintrust redaction", () => { "sensitive field value", ); } + expect(collectSensitiveValues(objectValue)).toContain("Nested2Pass"); + expect(collectSensitiveValues(arrayValue)).toContain("NestedToken"); }); test("redacts complex Playwright typing calls and rejects originals", () => { From 13a34ad634a66ce97e13a20becfaf731685d6bec Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Thu, 1 Oct 2026 01:21:13 +0000 Subject: [PATCH 12/14] Keep login snapshots publishable --- benchmarks/harbor/redact.ts | 47 +++++++++++++++++++++++++++-- benchmarks/harbor/results.test.ts | 50 +++++++++++++++++++++++++++++++ 2 files changed, 94 insertions(+), 3 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 479a5a32..436e5c9c 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -94,7 +94,9 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { let keyCursor = separator - 1; while (/\s/.test(text[keyCursor] ?? "")) keyCursor -= 1; - if (text[keyCursor] === '"' || text[keyCursor] === "'") { + const keyHasClosingQuote = + text[keyCursor] === '"' || text[keyCursor] === "'"; + if (keyHasClosingQuote) { keyCursor -= 1; while (text[keyCursor] === "\\") keyCursor -= 1; } @@ -105,7 +107,8 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { } keyCursor -= 1; } - const key = text.slice(keyCursor + 1, keyEnd); + const keyStart = keyCursor + 1; + const key = text.slice(keyStart, keyEnd); if (!key || !sensitiveField(key)) continue; let valueCursor = separator + 1; @@ -117,6 +120,17 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { let end = valueCursor; if (quote === '"' || quote === "'") { + let openingSlashCount = 0; + for (let cursor = keyStart - 2; text[cursor] === "\\"; cursor -= 1) { + openingSlashCount += 1; + } + if ( + !keyHasClosingQuote && + text[keyStart - 1] === quote && + openingSlashCount === slashCount + ) { + continue; + } start = valueCursor + slashCount + 1; end = text.length; for (let cursor = start; cursor < text.length; cursor += 1) { @@ -141,7 +155,19 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { end = structuredValueEnd(text, start); separator = end - 1; } else { - while (end < text.length && !/[\s,"'\}&;]/.test(text[end] ?? "")) { + while (end < text.length) { + const character = text[end] ?? ""; + if (/[\s,\}&;]/.test(character)) break; + if ( + (character === '"' || character === "'") && + !( + character === "'" && + /[A-Za-z0-9]/.test(text[end - 1] ?? "") && + /[A-Za-z0-9]/.test(text[end + 1] ?? "") + ) + ) { + break; + } end += 1; } separator = Math.max(separator, end - 1); @@ -481,6 +507,21 @@ function assertSensitiveAssignmentsRedacted(value: string): void { for (const match of value.matchAll(assignmentStart)) { if (!sensitiveField(match[1])) continue; const remainder = value.slice((match.index ?? 0) + match[0].length); + const valueQuote = remainder.match(/^(\\*)(["'])/); + const separator = Math.max( + match[0].lastIndexOf(":"), + match[0].lastIndexOf("="), + ); + if ( + valueQuote && + [...match[0].slice(0, separator)].filter( + (character) => character === valueQuote[2], + ).length % + 2 === + 1 + ) { + continue; + } const unquoted = remainder.replace(/^(?:\\*["'])?/, ""); if (!unquoted.startsWith(REDACTED)) { throw new Error( diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index f5cba6ba..7a694fa5 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -606,6 +606,38 @@ describe("Harbor result ingestion", () => { ]); }); + test("keeps login-page snapshots publishable", () => { + const root = fixture(); + const trajectoryPath = join( + root, + "task-one__abc", + "steps/run/agent/trajectory.json", + ); + const trajectory = JSON.parse(readFileSync(trajectoryPath, "utf8")) as { + steps: Array<{ + observation?: { results?: Array<{ content?: string }> }; + }>; + }; + trajectory.steps[2].observation!.results![0].content = JSON.stringify({ + success: true, + result: `- text: "Email:" +- textbox "Email:" +- text: "Password:" +Password: don't reuse one from another site`, + }); + writeJson(trajectoryPath, trajectory); + + const events = buildExperimentEvents( + [readBenchmarkArm({ name: "candidate", path: root })], + "login-snapshot", + ); + const published = JSON.stringify(events); + expect(published).toContain("Password:"); + expect(published).toContain("[REDACTED] reuse one from another site"); + expect(published).not.toContain("don't"); + expect(() => assertSafeToPublish(events)).not.toThrow(); + }); + test("harvests only sensitive values from private-info output", () => { const root = fixture(); writeJson(join(root, "task-one__abc", "steps/run/agent/trajectory.json"), { @@ -802,6 +834,24 @@ describe("Braintrust redaction", () => { expect(collectSensitiveValues(arrayValue)).toContain("NestedToken"); }); + test("keeps aria labels publishable and redacts prose values", () => { + const ariaSnapshot = `- text: "Email:" +- textbox "Email:" +- text: "Password:"`; + const escapedLabel = String.raw`- text: \"Password:\"`; + for (const label of [ariaSnapshot, escapedLabel]) { + expect(redactString(label)).toBe(label); + expect(collectSensitiveValues(label)).toEqual([]); + expect(() => assertSafeToPublish(label)).not.toThrow(); + } + + const prose = "Password: don't reuse one from another site"; + const redacted = redactString(prose); + expect(redacted).toBe("Password: [REDACTED] reuse one from another site"); + expect(collectSensitiveValues(prose)).toContain("don't"); + expect(() => assertSafeToPublish(redacted)).not.toThrow(); + }); + test("redacts complex Playwright typing calls and rejects originals", () => { const calls = [ `page.fill('#password', 'Str0ng)Pass!')`, From e4632ccdc1c82b913cb806578e96737ab443f960 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Thu, 1 Oct 2026 01:29:55 +0000 Subject: [PATCH 13/14] Redact values following snapshot labels --- benchmarks/harbor/redact.ts | 55 ++++++++++++++++++++++++++----- benchmarks/harbor/results.test.ts | 11 +++++++ 2 files changed, 58 insertions(+), 8 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 436e5c9c..15871a05 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -87,6 +87,17 @@ function structuredValueEnd(text: string, start: number): number { return text.length; } +function lineEndsAfter(text: string, start: number): boolean { + let cursor = start; + while (text[cursor] === " " || text[cursor] === "\t") cursor += 1; + return ( + cursor >= text.length || + text[cursor] === "\n" || + text[cursor] === "\r" || + (text[cursor] === "\\" && /[nr]/.test(text[cursor + 1] ?? "")) + ); +} + function sensitiveAssignments(text: string): SensitiveAssignment[] { const assignments: SensitiveAssignment[] = []; for (let separator = 0; separator < text.length; separator += 1) { @@ -115,7 +126,7 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { while (/\s/.test(text[valueCursor] ?? "")) valueCursor += 1; let slashCount = 0; while (text[valueCursor + slashCount] === "\\") slashCount += 1; - const quote = text[valueCursor + slashCount]; + let quote = text[valueCursor + slashCount]; let start = valueCursor; let end = valueCursor; @@ -124,13 +135,29 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { for (let cursor = keyStart - 2; text[cursor] === "\\"; cursor -= 1) { openingSlashCount += 1; } - if ( + const quotedLabel = !keyHasClosingQuote && text[keyStart - 1] === quote && - openingSlashCount === slashCount - ) { + openingSlashCount === slashCount; + if (quotedLabel && lineEndsAfter(text, valueCursor + slashCount + 1)) { continue; } + if (quotedLabel) { + valueCursor += slashCount + 1; + while (/\s/.test(text[valueCursor] ?? "")) valueCursor += 1; + if (text[valueCursor] === ":" || text[valueCursor] === "=") { + valueCursor += 1; + while (/\s/.test(text[valueCursor] ?? "")) valueCursor += 1; + } + slashCount = 0; + while (text[valueCursor + slashCount] === "\\") slashCount += 1; + quote = text[valueCursor + slashCount]; + start = valueCursor; + end = valueCursor; + } + } + + if (quote === '"' || quote === "'") { start = valueCursor + slashCount + 1; end = text.length; for (let cursor = start; cursor < text.length; cursor += 1) { @@ -157,7 +184,12 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { } else { while (end < text.length) { const character = text[end] ?? ""; - if (/[\s,\}&;]/.test(character)) break; + if ( + /[\s,\}&;]/.test(character) || + (character === "\\" && /[nrt]/.test(text[end + 1] ?? "")) + ) { + break; + } if ( (character === '"' || character === "'") && !( @@ -512,17 +544,24 @@ function assertSensitiveAssignmentsRedacted(value: string): void { match[0].lastIndexOf(":"), match[0].lastIndexOf("="), ); - if ( + const quotedLabel = valueQuote && [...match[0].slice(0, separator)].filter( (character) => character === valueQuote[2], ).length % 2 === - 1 + 1; + if ( + quotedLabel && + lineEndsAfter( + value, + (match.index ?? 0) + match[0].length + valueQuote[0].length, + ) ) { continue; } - const unquoted = remainder.replace(/^(?:\\*["'])?/, ""); + let unquoted = remainder.replace(/^(?:\\*["'])?/, ""); + if (quotedLabel) unquoted = unquoted.replace(/^\s*(?:[:=]\s*)?/, ""); if (!unquoted.startsWith(REDACTED)) { throw new Error( "Braintrust payload still contains a sensitive field value", diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 7a694fa5..95b53c6c 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -623,6 +623,7 @@ describe("Harbor result ingestion", () => { result: `- text: "Email:" - textbox "Email:" - text: "Password:" +- textbox "Password:": Filled2Pass Password: don't reuse one from another site`, }); writeJson(trajectoryPath, trajectory); @@ -634,6 +635,7 @@ Password: don't reuse one from another site`, const published = JSON.stringify(events); expect(published).toContain("Password:"); expect(published).toContain("[REDACTED] reuse one from another site"); + expect(published).not.toContain("Filled2Pass"); expect(published).not.toContain("don't"); expect(() => assertSafeToPublish(events)).not.toThrow(); }); @@ -850,6 +852,15 @@ describe("Braintrust redaction", () => { expect(redacted).toBe("Password: [REDACTED] reuse one from another site"); expect(collectSensitiveValues(prose)).toContain("don't"); expect(() => assertSafeToPublish(redacted)).not.toThrow(); + + const filledSnapshot = '- textbox "Password:": Filled2Pass'; + const redactedSnapshot = redactString(filledSnapshot); + expect(redactedSnapshot).not.toContain("Filled2Pass"); + expect(collectSensitiveValues(filledSnapshot)).toContain("Filled2Pass"); + expect(() => assertSafeToPublish(redactedSnapshot)).not.toThrow(); + expect(() => assertSafeToPublish(filledSnapshot)).toThrow( + "sensitive field value", + ); }); test("redacts complex Playwright typing calls and rejects originals", () => { From 4dda9e9b16e323a4c173a48068b0d2e351f95ac6 Mon Sep 17 00:00:00 2001 From: rgarcia <72655+rgarcia@users.noreply.github.com> Date: Thu, 1 Oct 2026 01:37:10 +0000 Subject: [PATCH 14/14] Skip snapshot attributes before redaction --- benchmarks/harbor/redact.ts | 40 +++++++++++++++++++++++++------ benchmarks/harbor/results.test.ts | 8 ++++--- 2 files changed, 38 insertions(+), 10 deletions(-) diff --git a/benchmarks/harbor/redact.ts b/benchmarks/harbor/redact.ts index 15871a05..6547773d 100644 --- a/benchmarks/harbor/redact.ts +++ b/benchmarks/harbor/redact.ts @@ -87,9 +87,14 @@ function structuredValueEnd(text: string, start: number): number { return text.length; } -function lineEndsAfter(text: string, start: number): boolean { +function skipHorizontalWhitespace(text: string, start: number): number { let cursor = start; while (text[cursor] === " " || text[cursor] === "\t") cursor += 1; + return cursor; +} + +function lineEndsAfter(text: string, start: number): boolean { + const cursor = skipHorizontalWhitespace(text, start); return ( cursor >= text.length || text[cursor] === "\n" || @@ -98,6 +103,14 @@ function lineEndsAfter(text: string, start: number): boolean { ); } +function skipSnapshotAttributes(text: string, start: number): number { + let cursor = skipHorizontalWhitespace(text, start); + while (text[cursor] === "[") { + cursor = skipHorizontalWhitespace(text, structuredValueEnd(text, cursor)); + } + return cursor; +} + function sensitiveAssignments(text: string): SensitiveAssignment[] { const assignments: SensitiveAssignment[] = []; for (let separator = 0; separator < text.length; separator += 1) { @@ -143,11 +156,13 @@ function sensitiveAssignments(text: string): SensitiveAssignment[] { continue; } if (quotedLabel) { - valueCursor += slashCount + 1; - while (/\s/.test(text[valueCursor] ?? "")) valueCursor += 1; + valueCursor = skipSnapshotAttributes( + text, + valueCursor + slashCount + 1, + ); + if (lineEndsAfter(text, valueCursor)) continue; if (text[valueCursor] === ":" || text[valueCursor] === "=") { - valueCursor += 1; - while (/\s/.test(text[valueCursor] ?? "")) valueCursor += 1; + valueCursor = skipHorizontalWhitespace(text, valueCursor + 1); } slashCount = 0; while (text[valueCursor + slashCount] === "\\") slashCount += 1; @@ -538,7 +553,8 @@ function assertSensitiveAssignmentsRedacted(value: string): void { /(?:^|\\[nrt]|[^A-Za-z0-9_-])(?:\\*["'])?([A-Za-z0-9_-]+)(?:\\*["'])?\s*[:=]\s*/gi; for (const match of value.matchAll(assignmentStart)) { if (!sensitiveField(match[1])) continue; - const remainder = value.slice((match.index ?? 0) + match[0].length); + const remainderStart = (match.index ?? 0) + match[0].length; + const remainder = value.slice(remainderStart); const valueQuote = remainder.match(/^(\\*)(["'])/); const separator = Math.max( match[0].lastIndexOf(":"), @@ -561,7 +577,17 @@ function assertSensitiveAssignmentsRedacted(value: string): void { continue; } let unquoted = remainder.replace(/^(?:\\*["'])?/, ""); - if (quotedLabel) unquoted = unquoted.replace(/^\s*(?:[:=]\s*)?/, ""); + if (quotedLabel && valueQuote) { + let valueStart = skipSnapshotAttributes( + value, + remainderStart + valueQuote[0].length, + ); + if (lineEndsAfter(value, valueStart)) continue; + if (value[valueStart] === ":" || value[valueStart] === "=") { + valueStart = skipHorizontalWhitespace(value, valueStart + 1); + } + unquoted = value.slice(valueStart).replace(/^(?:\\*["'])?/, ""); + } if (!unquoted.startsWith(REDACTED)) { throw new Error( "Braintrust payload still contains a sensitive field value", diff --git a/benchmarks/harbor/results.test.ts b/benchmarks/harbor/results.test.ts index 95b53c6c..1182ea84 100644 --- a/benchmarks/harbor/results.test.ts +++ b/benchmarks/harbor/results.test.ts @@ -623,7 +623,7 @@ describe("Harbor result ingestion", () => { result: `- text: "Email:" - textbox "Email:" - text: "Password:" -- textbox "Password:": Filled2Pass +- textbox "Password:" [disabled] [ref=e12]: Filled2Pass Password: don't reuse one from another site`, }); writeJson(trajectoryPath, trajectory); @@ -839,7 +839,8 @@ describe("Braintrust redaction", () => { test("keeps aria labels publishable and redacts prose values", () => { const ariaSnapshot = `- text: "Email:" - textbox "Email:" -- text: "Password:"`; +- text: "Password:" +- textbox "Password:" [disabled] [ref=e12]`; const escapedLabel = String.raw`- text: \"Password:\"`; for (const label of [ariaSnapshot, escapedLabel]) { expect(redactString(label)).toBe(label); @@ -853,7 +854,8 @@ describe("Braintrust redaction", () => { expect(collectSensitiveValues(prose)).toContain("don't"); expect(() => assertSafeToPublish(redacted)).not.toThrow(); - const filledSnapshot = '- textbox "Password:": Filled2Pass'; + const filledSnapshot = + '- textbox "Password:" [disabled] [ref=e12]: Filled2Pass'; const redactedSnapshot = redactString(filledSnapshot); expect(redactedSnapshot).not.toContain("Filled2Pass"); expect(collectSensitiveValues(filledSnapshot)).toContain("Filled2Pass");