diff --git a/.github/workflows/release-model-probe.yml b/.github/workflows/release-model-probe.yml index e9268fe3..1d85c946 100644 --- a/.github/workflows/release-model-probe.yml +++ b/.github/workflows/release-model-probe.yml @@ -11,6 +11,14 @@ on: description: Successful candidate Release workflow run required: true type: string + mode: + description: Raw Router diagnostic or installed Braid task + required: true + default: braid + type: choice + options: + - raw + - braid concurrency: group: braid-protected-account-proof @@ -68,10 +76,11 @@ jobs: cd candidate-source pnpm install --frozen-lockfile pnpm run build - - name: One bounded authenticated Gemini task from installed candidate + - name: Run one bounded Gemini probe env: BRAID_LIVE_TANGLE_ENV_JSON: ${{ secrets.BRAID_LIVE_TANGLE_ENV_JSON }} BRAID_PROBE_SOURCE: ${{ github.workspace }}/candidate-source BRAID_PROBE_ARTIFACT_ROOT: ${{ runner.temp }}/braid-model-probe BRAID_PROBE_COMMIT: ${{ inputs.commit }} + BRAID_PROBE_MODE: ${{ inputs.mode }} run: node scripts/release/model-probe.mjs diff --git a/scripts/release/model-probe.mjs b/scripts/release/model-probe.mjs index fa34fd2e..a6edef81 100644 --- a/scripts/release/model-probe.mjs +++ b/scripts/release/model-probe.mjs @@ -1,3 +1,4 @@ +import { randomUUID } from 'node:crypto' import { readFile, writeFile } from 'node:fs/promises' import { join, resolve } from 'node:path' @@ -13,6 +14,8 @@ import { const model = 'gemini-2.5-flash-lite' const marker = 'BRAID_GEMINI_PROBE_OK' +const mode = process.env.BRAID_PROBE_MODE ?? 'braid' +if (mode !== 'raw' && mode !== 'braid') throw new Error('Unknown probe mode') const source = resolve(process.env.BRAID_PROBE_SOURCE ?? '') const artifactRoot = resolve(process.env.BRAID_PROBE_ARTIFACT_ROOT ?? '') const commit = process.env.BRAID_PROBE_COMMIT @@ -53,11 +56,11 @@ const catalogResponse = await fetch(catalogUrl, { signal: AbortSignal.timeout(10_000), }) if (!catalogResponse.ok) { - throw new Error(`Protected credential model listing returned HTTP ${catalogResponse.status}`) + throw new Error(`Model catalog returned HTTP ${catalogResponse.status}`) } const catalog = await catalogResponse.json() if (!Array.isArray(catalog.data) || !catalog.data.some((item) => item?.id === model)) { - throw new Error('Protected credential catalog does not list the selected model') + throw new Error('Model catalog does not list the selected model') } const { identity } = await readCandidateIdentity({ repository: source, @@ -65,58 +68,115 @@ const { identity } = await readCandidateIdentity({ expectedCommit: commit, expectedVersion: '0.3.4', }) -const packed = await installPackedBraid(source, { - tarballPath: join(artifactRoot, identity.tarballPath), -}) -let config -let session -try { - if (packed.tarballSha256 !== identity.tarballSha256) { - throw new Error('Installed archive differs from the candidate') - } - config = await prepareProductionWorkspace({ - repository: source, - environment, - ...values, - model, +if (mode === 'raw') { + const response = await fetch(new URL('/v1/chat/completions', values.endpoint), { + method: 'POST', + headers: { + Authorization: `Bearer ${auth}`, + 'Content-Type': 'application/json', + 'Idempotency-Key': `braid-model-probe-${randomUUID()}`, + 'X-Tangle-Client': 'braid-release-probe/0.3.4', + }, + body: JSON.stringify({ + model, + messages: [ + { + role: 'system', + content: 'Follow the release prompt exactly and return the requested marker.', + }, + { role: 'user', content: `Reply with exactly ${marker}.` }, + ], + max_tokens: 32, + max_completion_tokens: 64, + reasoning_effort: 'none', + }), + signal: AbortSignal.timeout(30_000), }) - const profilePath = join(config.workspace, '.braid', 'profiles', 'profile-tangle-inference.json') - const profile = JSON.parse(await readFile(profilePath, 'utf8')) - profile.model.maxVisibleOutputTokens = 32 - profile.model.maxTotalOutputTokens = 64 - await writeFile(profilePath, `${JSON.stringify(profile, null, 2)}\n`, { mode: 0o600 }) - process.env.BRAID_LIVE_REQUIRED_TIMEOUT_MS = '60000' - const turn = await runHeadlessTurn({ - binary: packed.binary, - config, - marker, - prompt: `Reply with exactly ${marker}.`, - }) - session = turn.session - const receipt = { - schema: 'braid.release-model-probe.v1', + const body = await response.json().catch(() => null) + const costHeader = Number(response.headers.get('x-tangle-cost-usd')) + const costUsd = + Number.isFinite(costHeader) && costHeader >= 0 && response.headers.has('x-tangle-cost-usd') + ? costHeader + : null + const diagnostic = { + schema: 'braid.release-model-raw-probe.v1', commit, - version: identity.braidVersion, - tarballSha256: packed.tarballSha256, - model: turn.run.model ?? model, - status: turn.run.status, - runId: turn.run.id, - materializationDigest: turn.run.materializationDigest, - costStatus: turn.run.costStatus ?? 'unknown', - costUsd: turn.run.costUsd ?? null, + tarballSha256: identity.tarballSha256, + requestedModel: model, + httpStatus: response.status, + generationId: response.headers.get('x-generation-id'), + servedModel: response.headers.get('x-tangle-served-model') ?? body?.model ?? null, + costUsd, + promptTokens: body?.usage?.prompt_tokens ?? null, + completionTokens: body?.usage?.completion_tokens ?? null, + errorCode: body?.error?.code ?? null, + errorMessage: response.ok + ? null + : safeMessage(body?.error?.message ?? 'unknown Router error', environment), maxVisibleOutputTokens: 32, maxTotalOutputTokens: 64, - markerMatched: true, } - process.stdout.write(`${JSON.stringify(receipt)}\n`) - if (typeof receipt.costUsd === 'number' && receipt.costUsd > 0.01) { - throw new Error('One-turn cost exceeded the $0.01 probe ceiling') + process.stdout.write(`${JSON.stringify(diagnostic)}\n`) + if (!response.ok || (costUsd !== null && costUsd > 0.01)) process.exitCode = 1 +} else { + const packed = await installPackedBraid(source, { + tarballPath: join(artifactRoot, identity.tarballPath), + }) + let config + let session + try { + if (packed.tarballSha256 !== identity.tarballSha256) { + throw new Error('Installed archive differs from the candidate') + } + config = await prepareProductionWorkspace({ + repository: source, + environment, + ...values, + model, + }) + const profilePath = join( + config.workspace, + '.braid', + 'profiles', + 'profile-tangle-inference.json', + ) + const profile = JSON.parse(await readFile(profilePath, 'utf8')) + profile.model.maxVisibleOutputTokens = 32 + profile.model.maxTotalOutputTokens = 64 + await writeFile(profilePath, `${JSON.stringify(profile, null, 2)}\n`, { mode: 0o600 }) + process.env.BRAID_LIVE_REQUIRED_TIMEOUT_MS = '60000' + const turn = await runHeadlessTurn({ + binary: packed.binary, + config, + marker, + prompt: `Reply with exactly ${marker}.`, + }) + session = turn.session + const receipt = { + schema: 'braid.release-model-probe.v1', + commit, + version: identity.braidVersion, + tarballSha256: packed.tarballSha256, + model: turn.run.model ?? model, + status: turn.run.status, + runId: turn.run.id, + materializationDigest: turn.run.materializationDigest, + costStatus: turn.run.costStatus ?? 'unknown', + costUsd: turn.run.costUsd ?? null, + maxVisibleOutputTokens: 32, + maxTotalOutputTokens: 64, + markerMatched: true, + } + process.stdout.write(`${JSON.stringify(receipt)}\n`) + if (typeof receipt.costUsd === 'number' && receipt.costUsd > 0.01) { + throw new Error('One-turn cost exceeded the $0.01 probe ceiling') + } + } catch (error) { + process.stderr.write(`Protected model probe failed: ${safeMessage(error, environment)}\n`) + process.exitCode = 1 + } finally { + if (session) await closeSession(session) + if (config) await config.cleanup() + await packed.cleanup() } -} catch (error) { - process.stderr.write(`Protected model probe failed: ${safeMessage(error, environment)}\n`) - process.exitCode = 1 -} finally { - if (session) await closeSession(session) - if (config) await config.cleanup() - await packed.cleanup() }