From 393e4cd6f93b7ecb6688aa1ccc437326be6ba24e Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Tue, 6 Oct 2026 11:14:29 -0700 Subject: [PATCH 1/5] fix(ci): start each HTTP end-to-end app from an empty Turbopack dev cache The stop-after suite failed 12 of 175 CI runs (Oct 1-6), always in its own `next dev` app, which restored the Turbopack dev cache (.next/dev) the SCIM/version-compare app left in the same job. That app ran with other NEXT_PUBLIC_* values, and `next dev` SIGKILLs its server 100ms after SIGTERM. Restoring the cache: - panicked Turbopack at startup (inner_of_upper_lost_follower), 6 runs; - panicked mid-compile of the execute route ("socket connection was closed unexpectedly"), 4 runs; - wedged that compile until the 300s request timeout ("The operation timed out."), 2 runs. No run reached the executor: the cleanup check found no open execution log. - Each app step removes .next/dev before starting, so no app restores another app's cache. - A failing step prints the server log tail, not only a startup failure. - The suite compiles the execute route with a refused request under its own 300s budget and named check; every later request and CLI run is bounded at 60s. A timed-out or dropped request names the route and elapsed time, lands in the report with a null status, and a CLI timeout is reported as one. --- .github/workflows/test-build.yml | 14 +- .../scripts/test-workflow-stop-after-e2e.ts | 170 +++++++++++------- 2 files changed, 121 insertions(+), 63 deletions(-) diff --git a/.github/workflows/test-build.yml b/.github/workflows/test-build.yml index 5e6612d404a..9fe9439609f 100644 --- a/.github/workflows/test-build.yml +++ b/.github/workflows/test-build.yml @@ -232,17 +232,19 @@ jobs: report_dir="$RUNNER_TEMP/e2e" server_log="$report_dir/scim-next.log" mkdir -p "$report_dir" + rm -rf .next/dev node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3017 > "$server_log" 2>&1 & server_pid=$! finish() { + status=$? kill "$server_pid" 2>/dev/null || true wait "$server_pid" 2>/dev/null || true awk '/^ (GET|POST|PUT|PATCH|DELETE|HEAD) \/api\// { print }' "$server_log" > "$report_dir/scim-http-status.log" + [ "$status" -eq 0 ] || tail -n 200 "$server_log" } trap finish EXIT fail_startup() { echo "::error::$1" - tail -n 200 "$server_log" exit 1 } started=$SECONDS @@ -267,6 +269,12 @@ jobs: # A self-hosted app: hosted billing admits a run only through a Redis usage # reservation, and the SCIM suite above asserts PostgreSQL rate-limit storage, # so workflow execution gets its own app rather than adding Redis to that one. + # + # Every app in this job starts from an empty Turbopack dev cache (.next/dev). + # Restoring the one the previous app left, built with other NEXT_PUBLIC_* + # values and cut off by `next dev` killing its server 100ms after SIGTERM, + # panicked Turbopack ("inner_of_upper_lost_follower") or wedged the first + # route compile in 12 of 175 CI runs. - name: Verify single-block workflow runs over real HTTP working-directory: apps/sim env: @@ -283,17 +291,19 @@ jobs: report_dir="$RUNNER_TEMP/e2e" server_log="$report_dir/stop-after-next.log" mkdir -p "$report_dir" + rm -rf .next/dev node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3018 > "$server_log" 2>&1 & server_pid=$! finish() { + status=$? kill "$server_pid" 2>/dev/null || true wait "$server_pid" 2>/dev/null || true awk '/^ (GET|POST|PUT|PATCH|DELETE|HEAD) \/api\// { print }' "$server_log" > "$report_dir/stop-after-http-status.log" + [ "$status" -eq 0 ] || tail -n 200 "$server_log" } trap finish EXIT fail_startup() { echo "::error::$1" - tail -n 200 "$server_log" exit 1 } started=$SECONDS diff --git a/apps/sim/scripts/test-workflow-stop-after-e2e.ts b/apps/sim/scripts/test-workflow-stop-after-e2e.ts index c27ab483826..e465ee02247 100644 --- a/apps/sim/scripts/test-workflow-stop-after-e2e.ts +++ b/apps/sim/scripts/test-workflow-stop-after-e2e.ts @@ -37,8 +37,10 @@ import { readResponseTextWithLimit } from '@/lib/core/utils/stream-limits' const logger = createLogger('WorkflowStopAfterE2E') const execFileAsync = promisify(execFile) const MAX_RESPONSE_BYTES = 2 * 1024 * 1024 -/** The first execute request cold-compiles the route under `next dev`. */ -const REQUEST_TIMEOUT_MS = 300_000 +/** The first execute request cold-compiles the route's module graph under `next dev`. */ +const ROUTE_COMPILE_TIMEOUT_MS = 300_000 +/** Every later request hits the compiled route; the slowest fixture run waits about `SLOW_MS`. */ +const REQUEST_TIMEOUT_MS = 60_000 const SLOW_SECONDS = 4 const SLOW_MS = SLOW_SECONDS * 1000 const startedAt = new Date().toISOString() @@ -67,7 +69,8 @@ const personalKey = `sk-sim-fixture-${generateId()}` const cliPath = fileURLToPath(new URL('../../../packages/sim-cli/src/index.ts', import.meta.url)) const checks: { name: string; status: 'passed' | 'failed'; durationMs: number; error?: string }[] = [] -const requests: { method: string; path: string; status: number; durationMs: number }[] = [] +/** `status` is null when the request ended without a complete response. */ +const requests: { method: string; path: string; status: number | null; durationMs: number }[] = [] let directory: string | undefined interface PipelineFixture { @@ -225,37 +228,48 @@ async function seed() { }) } +function isTimeout(error: unknown): boolean { + return error instanceof DOMException && error.name === 'TimeoutError' +} + async function execute( workflowId: string, body: V2ExecuteWorkflowBody, - expectedStatus = 200 + { expectedStatus = 200, timeoutMs = REQUEST_TIMEOUT_MS } = {} ): Promise> { const url = new URL(`/api/v2/workflows/${workflowId}/execute`, baseUrl) const started = performance.now() - // boundary-raw-fetch: protocol E2E exercises a separately running local app over real HTTP - const response = await fetch(url, { - method: 'POST', - redirect: 'error', - signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS), - headers: { - accept: 'application/json', - 'content-type': 'application/json', - 'x-api-key': personalKey, - 'x-forwarded-for': '127.0.0.1', - }, - body: JSON.stringify(body), - }) - requests.push({ - method: 'POST', - path: url.pathname, - status: response.status, - durationMs: Math.round(performance.now() - started), - }) - const text = await readResponseTextWithLimit(response, { - maxBytes: MAX_RESPONSE_BYTES, - label: 'Stop-after E2E response', - }) - assert.equal(response.status, expectedStatus, `${url.pathname}: ${truncate(text, 500)}`) + const elapsed = () => Math.round(performance.now() - started) + let status: number | null = null + let text: string + try { + // boundary-raw-fetch: protocol E2E exercises a separately running local app over real HTTP + const response = await fetch(url, { + method: 'POST', + redirect: 'error', + signal: AbortSignal.timeout(timeoutMs), + headers: { + accept: 'application/json', + 'content-type': 'application/json', + 'x-api-key': personalKey, + 'x-forwarded-for': '127.0.0.1', + }, + body: JSON.stringify(body), + }) + status = response.status + text = await readResponseTextWithLimit(response, { + maxBytes: MAX_RESPONSE_BYTES, + label: 'Stop-after E2E response', + }) + } catch (error) { + const reason = isTimeout(error) + ? `no complete response within ${timeoutMs / 1000}s` + : getErrorMessage(error) + throw new Error(`POST ${url.pathname} failed after ${elapsed()} ms: ${reason}`) + } finally { + requests.push({ method: 'POST', path: url.pathname, status, durationMs: elapsed() }) + } + assert.equal(status, expectedStatus, `${url.pathname}: ${truncate(text, 500)}`) return record(JSON.parse(text)) } @@ -268,49 +282,74 @@ async function run( return data } -async function expectBadRequest(workflowId: string, body: V2ExecuteWorkflowBody, code: string) { - const status = code === 'NOT_FOUND' ? 404 : 400 - const error = record((await execute(workflowId, body, status)).error) +async function expectBadRequest( + workflowId: string, + body: V2ExecuteWorkflowBody, + code: string, + timeoutMs = REQUEST_TIMEOUT_MS +) { + const expectedStatus = code === 'NOT_FOUND' ? 404 : 400 + const error = record((await execute(workflowId, body, { expectedStatus, timeoutMs })).error) assert.equal(error.code, code) } -async function runCli(args: string[]): Promise { +/** Runs the CLI; a run it must fail exits non-zero and still prints the run on stdout. */ +async function execCli( + args: string[] +): Promise<{ exitCode: number; stdout: string; stderr: string }> { assert(directory, 'CLI fixture directory must exist') - const { stdout } = await execFileAsync( - 'bun', - [ - '--no-env-file', - cliPath, - '--endpoint', - baseUrl.origin, - '--workspace', - workspaceId, - '--output', - 'json', - 'workflows', - 'run', - ...args, - ], - { - cwd: directory, - env: { ...process.env, SIM_CONFIG_DIR: directory, SIM_API_KEY: personalKey, NO_COLOR: '1' }, - timeout: REQUEST_TIMEOUT_MS, - maxBuffer: MAX_RESPONSE_BYTES, - } + try { + const { stdout, stderr } = await execFileAsync( + 'bun', + [ + '--no-env-file', + cliPath, + '--endpoint', + baseUrl.origin, + '--workspace', + workspaceId, + '--output', + 'json', + 'workflows', + 'run', + ...args, + ], + { + cwd: directory, + env: { ...process.env, SIM_CONFIG_DIR: directory, SIM_API_KEY: personalKey, NO_COLOR: '1' }, + timeout: REQUEST_TIMEOUT_MS, + maxBuffer: MAX_RESPONSE_BYTES, + } + ) + return { exitCode: 0, stdout, stderr } + } catch (error) { + assert(isRecordLike(error), getErrorMessage(error)) + assert(!error.killed, `sim workflows run did not exit within ${REQUEST_TIMEOUT_MS / 1000}s`) + assert( + typeof error.code === 'number' && + typeof error.stdout === 'string' && + typeof error.stderr === 'string', + getErrorMessage(error) + ) + return { exitCode: error.code, stdout: error.stdout, stderr: error.stderr } + } +} + +async function runCli(args: string[]): Promise { + const { exitCode, stdout, stderr } = await execCli(args) + assert.equal( + exitCode, + 0, + `sim workflows run exited ${exitCode}: ${truncate(stderr || stdout, 500)}` ) return v2ExecuteWorkflowDataSchema.parse(JSON.parse(stdout)) } /** A CLI run the command itself must fail: exits non-zero and prints the failed run. */ async function runCliExpectingFailure(args: string[]): Promise { - try { - await runCli(args) - } catch (error) { - assert(isRecordLike(error) && typeof error.stdout === 'string', getErrorMessage(error)) - assert.notEqual(error.code, 0, 'a failed run must exit non-zero') - return v2ExecuteWorkflowDataSchema.parse(JSON.parse(error.stdout)) - } - assert.fail('the CLI exited 0 for a run that must fail') + const { exitCode, stdout } = await execCli(args) + assert.notEqual(exitCode, 0, 'the CLI exited 0 for a run that must fail') + return v2ExecuteWorkflowDataSchema.parse(JSON.parse(stdout)) } const selectAll = ['Slow.status', 'Check.status', 'After.status'] @@ -318,6 +357,15 @@ const selectAll = ['Slow.status', 'Check.status', 'After.status'] try { await check('seed disposable workspace, personal key and fixtures', seed) + await check('the execute route compiles and refuses a run before it starts', () => + expectBadRequest( + pipeline.workflowId, + { run: { source: 'manual', stopAfterBlockId: '' } }, + 'BAD_REQUEST', + ROUTE_COMPILE_TIMEOUT_MS + ) + ) + let sourceRunId = '' await check('a full manual run executes every block and persists its state', async () => { const full = await run(pipeline.workflowId, { From 1f3ea39b21bc361f324ba685349161a3920f149f Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Tue, 6 Oct 2026 11:26:03 -0700 Subject: [PATCH 2/5] fix(ci): start the desktop inbox app from an empty Turbopack dev cache too --- .github/workflows/test-build.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.github/workflows/test-build.yml b/.github/workflows/test-build.yml index 9fe9439609f..6acfe3f697a 100644 --- a/.github/workflows/test-build.yml +++ b/.github/workflows/test-build.yml @@ -339,17 +339,19 @@ jobs: report_dir="$RUNNER_TEMP/e2e" server_log="$report_dir/desktop-inbox-next.log" mkdir -p "$report_dir" + rm -rf .next/dev node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3019 > "$server_log" 2>&1 & server_pid=$! finish() { + status=$? kill "$server_pid" 2>/dev/null || true wait "$server_pid" 2>/dev/null || true awk '/^ (GET|POST|PUT|PATCH|DELETE|HEAD) \/api\// { print }' "$server_log" > "$report_dir/desktop-inbox-http-status.log" + [ "$status" -eq 0 ] || tail -n 200 "$server_log" } trap finish EXIT fail_startup() { echo "::error::$1" - tail -n 200 "$server_log" exit 1 } started=$SECONDS From 6583e33259b16f93fee7b1db10649e6db39c24c4 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Tue, 6 Oct 2026 11:26:22 -0700 Subject: [PATCH 3/5] fix(e2e): record a status only for a fully read stop-after response --- apps/sim/scripts/test-workflow-stop-after-e2e.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/apps/sim/scripts/test-workflow-stop-after-e2e.ts b/apps/sim/scripts/test-workflow-stop-after-e2e.ts index e465ee02247..6088f489acd 100644 --- a/apps/sim/scripts/test-workflow-stop-after-e2e.ts +++ b/apps/sim/scripts/test-workflow-stop-after-e2e.ts @@ -256,11 +256,11 @@ async function execute( }, body: JSON.stringify(body), }) - status = response.status text = await readResponseTextWithLimit(response, { maxBytes: MAX_RESPONSE_BYTES, label: 'Stop-after E2E response', }) + status = response.status } catch (error) { const reason = isTimeout(error) ? `no complete response within ${timeoutMs / 1000}s` From edc1d47aadcab7d23aad8d0248861bebc9da579e Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Tue, 6 Oct 2026 12:42:39 -0700 Subject: [PATCH 4/5] fix(ci): stop each HTTP end-to-end app's whole process session before the next one starts --- .github/scripts/stop-session.sh | 43 ++++++++++++++++++++++++++++++++ .github/workflows/test-build.yml | 37 ++++++++++++++++----------- 2 files changed, 66 insertions(+), 14 deletions(-) create mode 100755 .github/scripts/stop-session.sh diff --git a/.github/scripts/stop-session.sh b/.github/scripts/stop-session.sh new file mode 100755 index 00000000000..860561a568e --- /dev/null +++ b/.github/scripts/stop-session.sh @@ -0,0 +1,43 @@ +#!/usr/bin/env bash +# Stops every process in a session started with `setsid`, and returns only once none is left. +# +# Usage: stop-session.sh +# +# `next dev` runs its server and workers as child processes. Waiting on `next dev` alone returns +# while those can still be running, and the next app in the job reuses the same .next directory. +# The leader stays a zombie until the shell that started it waits on it, so zombies don't count. +set -u + +leader=$1 + +running() { + ps -s "$leader" -o pid=,stat= 2>/dev/null | awk '$2 !~ /^Z/ { found = 1 } END { exit !found }' +} + +leader_running() { + ps -p "$leader" -o stat= 2>/dev/null | grep -qv '^Z' +} + +wait_for() { + for _ in $(seq 1 100); do + "$@" || return 0 + sleep 0.1 + done + return 1 +} + +kill -TERM -- "-$leader" 2>/dev/null || true +wait_for leader_running +if running; then + echo "Processes from session $leader outlived its leader:" + ps -s "$leader" -o pid,stat,etimes,args 2>/dev/null | cut -c1-200 || true +fi +wait_for running && exit 0 + +echo "::warning::Processes from session $leader were still running 10s after SIGTERM:" +ps -s "$leader" -o pid,stat,etimes,args 2>/dev/null || true +kill -KILL -- "-$leader" 2>/dev/null || true +wait_for running && exit 0 + +echo "::error::Processes from session $leader survived SIGKILL." +exit 1 diff --git a/.github/workflows/test-build.yml b/.github/workflows/test-build.yml index 6acfe3f697a..e61fb6e9f7d 100644 --- a/.github/workflows/test-build.yml +++ b/.github/workflows/test-build.yml @@ -233,14 +233,17 @@ jobs: server_log="$report_dir/scim-next.log" mkdir -p "$report_dir" rm -rf .next/dev - node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3017 > "$server_log" 2>&1 & + setsid node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3017 > "$server_log" 2>&1 & server_pid=$! finish() { status=$? - kill "$server_pid" 2>/dev/null || true + bash ../../.github/scripts/stop-session.sh "$server_pid" || status=1 wait "$server_pid" 2>/dev/null || true awk '/^ (GET|POST|PUT|PATCH|DELETE|HEAD) \/api\// { print }' "$server_log" > "$report_dir/scim-http-status.log" - [ "$status" -eq 0 ] || tail -n 200 "$server_log" + if [ "$status" -ne 0 ]; then + tail -n 200 "$server_log" + exit "$status" + fi } trap finish EXIT fail_startup() { @@ -270,11 +273,11 @@ jobs: # reservation, and the SCIM suite above asserts PostgreSQL rate-limit storage, # so workflow execution gets its own app rather than adding Redis to that one. # - # Every app in this job starts from an empty Turbopack dev cache (.next/dev). - # Restoring the one the previous app left, built with other NEXT_PUBLIC_* - # values and cut off by `next dev` killing its server 100ms after SIGTERM, - # panicked Turbopack ("inner_of_upper_lost_follower") or wedged the first - # route compile in 12 of 175 CI runs. + # The apps in this job share apps/sim/.next. Each starts from an empty Turbopack + # dev cache: one written under other NEXT_PUBLIC_* values, by a server `next dev` + # SIGKILLs 100ms after SIGTERM, can panic Turbopack or wedge a route compile on + # restore. Each runs in its own session, and stop-session.sh returns only once + # all of it has exited, so no app starts beside one still writing .next/dev. - name: Verify single-block workflow runs over real HTTP working-directory: apps/sim env: @@ -292,14 +295,17 @@ jobs: server_log="$report_dir/stop-after-next.log" mkdir -p "$report_dir" rm -rf .next/dev - node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3018 > "$server_log" 2>&1 & + setsid node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3018 > "$server_log" 2>&1 & server_pid=$! finish() { status=$? - kill "$server_pid" 2>/dev/null || true + bash ../../.github/scripts/stop-session.sh "$server_pid" || status=1 wait "$server_pid" 2>/dev/null || true awk '/^ (GET|POST|PUT|PATCH|DELETE|HEAD) \/api\// { print }' "$server_log" > "$report_dir/stop-after-http-status.log" - [ "$status" -eq 0 ] || tail -n 200 "$server_log" + if [ "$status" -ne 0 ]; then + tail -n 200 "$server_log" + exit "$status" + fi } trap finish EXIT fail_startup() { @@ -340,14 +346,17 @@ jobs: server_log="$report_dir/desktop-inbox-next.log" mkdir -p "$report_dir" rm -rf .next/dev - node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3019 > "$server_log" 2>&1 & + setsid node ../../node_modules/next/dist/bin/next dev --hostname 127.0.0.1 --port 3019 > "$server_log" 2>&1 & server_pid=$! finish() { status=$? - kill "$server_pid" 2>/dev/null || true + bash ../../.github/scripts/stop-session.sh "$server_pid" || status=1 wait "$server_pid" 2>/dev/null || true awk '/^ (GET|POST|PUT|PATCH|DELETE|HEAD) \/api\// { print }' "$server_log" > "$report_dir/desktop-inbox-http-status.log" - [ "$status" -eq 0 ] || tail -n 200 "$server_log" + if [ "$status" -ne 0 ]; then + tail -n 200 "$server_log" + exit "$status" + fi } trap finish EXIT fail_startup() { From bd1e89dc5fe25c21de9d02a9cfd25f2e0f846b19 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Tue, 6 Oct 2026 13:20:39 -0700 Subject: [PATCH 5/5] fix(e2e): compile the desktop executor routes before timing executor calls --- apps/sim/scripts/test-desktop-inbox-e2e.ts | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/apps/sim/scripts/test-desktop-inbox-e2e.ts b/apps/sim/scripts/test-desktop-inbox-e2e.ts index 0becab4ab71..5ef02e09210 100644 --- a/apps/sim/scripts/test-desktop-inbox-e2e.ts +++ b/apps/sim/scripts/test-desktop-inbox-e2e.ts @@ -413,6 +413,17 @@ async function run() { assert(registration.reconcileMs < PICKUP_GRACE_SECONDS * 1000) }) + /** `next dev` compiles a route on its first request, which must not count against the timed checks. */ + await check('refuses malformed claim, lease and completion bodies', async () => { + for (const path of [ + '/api/desktop/tool/claim', + '/api/desktop/tool/lease', + '/api/desktop/tool/complete', + ]) { + await request(desktop, 'POST', path, { body: {}, expected: 400 }) + } + }) + /** Starts absent, so only the stream open below can mark the device present. */ await redis.del(`desktop:presence:${desktop.deviceId}`) const doorbell = openDoorbell(desktop)