From 7aa799d20bfaa5ee4e13577a128b202aae1fc0e9 Mon Sep 17 00:00:00 2001 From: dapiced Date: Tue, 29 Sep 2026 18:21:06 -0400 Subject: [PATCH] fix(deploy): survive Hugging Face 429 - retry with backoff, cache the sharded model, optional HF_TOKEN Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/workflows/ci.yml | 13 +++++++ .github/workflows/llm-bench.yml | 2 ++ scripts/fetch-retry.d.mts | 19 +++++++++++ scripts/fetch-retry.mjs | 54 +++++++++++++++++++++++++++++ scripts/prepare-llm.mjs | 20 +++++++++-- src/lib/fetchRetry.test.ts | 60 +++++++++++++++++++++++++++++++++ 6 files changed, 165 insertions(+), 3 deletions(-) create mode 100644 scripts/fetch-retry.d.mts create mode 100644 scripts/fetch-retry.mjs create mode 100644 src/lib/fetchRetry.test.ts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9f0bc0d..cb585ab 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -127,7 +127,20 @@ jobs: # checkout, for weights only the opt-in chat engine downloads. The script # verifies each file's size and fails loudly rather than shipping a # truncated model. + # V41.1 — Hugging Face rate-limits shared runner IPs (HTTP 429). The + # sharded model is cached, keyed on the script that pins it, so a normal + # deploy never touches Hugging Face; a miss retries with backoff and uses + # the optional HF_TOKEN secret to lift the anonymous limit. + - name: Cache the sharded language model + id: llm + uses: actions/cache@v4 + with: + path: dist/llm + key: llm-sharded-${{ hashFiles('scripts/prepare-llm.mjs') }} - name: Fetch and shard the local language model + if: steps.llm.outputs.cache-hit != 'true' + env: + HF_TOKEN: ${{ secrets.HF_TOKEN }} run: npm run llm:prepare -- dist/llm - name: Ensure the Pages project exists run: npx wrangler pages project create labml --production-branch=main || true diff --git a/.github/workflows/llm-bench.yml b/.github/workflows/llm-bench.yml index 3c460fa..b2c9938 100644 --- a/.github/workflows/llm-bench.yml +++ b/.github/workflows/llm-bench.yml @@ -47,6 +47,8 @@ jobs: path: .llm-cache key: llm-cache-qwen3-0.6b-dq-q4f16-v1 - if: steps.weights.outputs.cache-hit != 'true' + env: + HF_TOKEN: ${{ secrets.HF_TOKEN }} run: npm run llm:fetch # Constrained decoding is the default, shared with the browser through # DECODE_CONSTRAINED_BY_DEFAULT — the first run of this workflow measured diff --git a/scripts/fetch-retry.d.mts b/scripts/fetch-retry.d.mts new file mode 100644 index 0000000..60763cb --- /dev/null +++ b/scripts/fetch-retry.d.mts @@ -0,0 +1,19 @@ +export interface RetryOptions { + fetchFn?: (url: string, init?: RequestInit) => Promise; + sleep?: (ms: number) => Promise; + attempts?: number; + maxDelayMs?: number; + onRetry?: (info: { attempt: number; delay: number; reason: string }) => void; +} + +export function isRetryable(status: number): boolean; +export function retryDelayMs( + response: Response | undefined, + attempt: number, + maxDelayMs: number, +): number; +export function fetchWithRetry( + url: string, + init?: RequestInit, + options?: RetryOptions, +): Promise; diff --git a/scripts/fetch-retry.mjs b/scripts/fetch-retry.mjs new file mode 100644 index 0000000..d6eface --- /dev/null +++ b/scripts/fetch-retry.mjs @@ -0,0 +1,54 @@ +/** + * V41.1 — Hugging Face answers anonymous CI runners with 429 when too many + * requests share an IP. The deploy used to die on the first one; it now waits + * (honouring Retry-After when given) and tries again. 404 and other 4xx are + * returned untouched: retrying them would only hide a real mistake. + */ +const sleepMs = (ms) => new Promise((resolve) => setTimeout(resolve, ms)); + +export function isRetryable(status) { + return status === 408 || status === 429 || status >= 500; +} + +export function retryDelayMs(response, attempt, maxDelayMs) { + const header = response?.headers?.get?.('retry-after'); + if (header) { + const seconds = Number(header); + if (Number.isFinite(seconds)) return Math.min(seconds * 1000, maxDelayMs); + const at = Date.parse(header); + if (Number.isFinite(at)) return Math.min(Math.max(at - Date.now(), 0), maxDelayMs); + } + const base = 2000 * 2 ** attempt; + return Math.min(base + Math.floor(Math.random() * 1000), maxDelayMs); +} + +export async function fetchWithRetry(url, init = {}, options = {}) { + const { + fetchFn = fetch, + sleep = sleepMs, + attempts = 6, + maxDelayMs = 60_000, + onRetry = () => {}, + } = options; + let lastError; + for (let attempt = 0; attempt < attempts; attempt++) { + const last = attempt === attempts - 1; + let response; + try { + response = await fetchFn(url, init); + } catch (error) { + lastError = error; + if (last) throw error; + const delay = retryDelayMs(undefined, attempt, maxDelayMs); + onRetry({ attempt: attempt + 1, delay, reason: String(error?.message ?? error) }); + await sleep(delay); + continue; + } + if (response.ok || !isRetryable(response.status) || last) return response; + const delay = retryDelayMs(response, attempt, maxDelayMs); + onRetry({ attempt: attempt + 1, delay, reason: `HTTP ${response.status}` }); + await response.body?.cancel?.().catch(() => {}); + await sleep(delay); + } + throw lastError; +} diff --git a/scripts/prepare-llm.mjs b/scripts/prepare-llm.mjs index 66ffff9..15fc09b 100644 --- a/scripts/prepare-llm.mjs +++ b/scripts/prepare-llm.mjs @@ -21,6 +21,7 @@ */ import { mkdir, writeFile } from 'node:fs/promises'; import { dirname, join } from 'node:path'; +import { fetchWithRetry } from './fetch-retry.mjs'; const REPO = 'onnx-community/Qwen3-0.6B-DQ-ONNX'; /** Pinned so a silent upstream change can never reach production unnoticed. */ @@ -60,9 +61,21 @@ function url(path) { async function download(path) { // Hugging Face refuses requests without a User-Agent behind some proxies. - const response = await fetch(url(path), { - headers: { 'User-Agent': 'LabML-build/1.0 (+https://app.dominicdapice.com)' }, - }); + const headers = { 'User-Agent': 'LabML-build/1.0 (+https://app.dominicdapice.com)' }; + // An HF token (optional) lifts the anonymous rate limit that CI runners hit. + if (process.env.HF_TOKEN && !process.env.LLM_MIRROR) { + headers.Authorization = `Bearer ${process.env.HF_TOKEN}`; + } + const response = await fetchWithRetry( + url(path), + { headers }, + { + onRetry: ({ attempt, delay, reason }) => + console.log( + ` ${path}: ${reason}, nouvel essai ${attempt} dans ${Math.round(delay / 1000)} s`, + ), + }, + ); if (!response.ok) throw new Error(`fetch-failed:${path}:${response.status}`); const bytes = new Uint8Array(await response.arrayBuffer()); const expected = FILES[path]; @@ -87,6 +100,7 @@ async function main() { let totalBytes = 0; for (const path of [...Object.keys(FILES), 'LICENSE']) { + if (files.length > 0 || totalBytes > 0) await new Promise((r) => setTimeout(r, 500)); const bytes = await download(path); totalBytes += bytes.byteLength; if (flat || bytes.byteLength <= SHARD_BYTES) { diff --git a/src/lib/fetchRetry.test.ts b/src/lib/fetchRetry.test.ts new file mode 100644 index 0000000..14628c4 --- /dev/null +++ b/src/lib/fetchRetry.test.ts @@ -0,0 +1,60 @@ +import { describe, expect, it, vi } from 'vitest'; +import { fetchWithRetry, retryDelayMs } from '../../scripts/fetch-retry.mjs'; + +const ok = () => new Response('ok', { status: 200 }); +const status = (code: number, headers: Record = {}) => + new Response('', { status: code, headers }); + +describe('fetchWithRetry', () => { + it('retries a 429 and returns the eventual success', async () => { + const fetchFn = vi.fn().mockResolvedValueOnce(status(429)).mockResolvedValueOnce(ok()); + const sleep = vi.fn().mockResolvedValue(undefined); + const response = await fetchWithRetry('u', {}, { fetchFn, sleep, attempts: 3 }); + expect(response.status).toBe(200); + expect(fetchFn).toHaveBeenCalledTimes(2); + expect(sleep).toHaveBeenCalledTimes(1); + }); + + it('retries 5xx and network errors', async () => { + const fetchFn = vi + .fn() + .mockRejectedValueOnce(new Error('ECONNRESET')) + .mockResolvedValueOnce(status(503)) + .mockResolvedValueOnce(ok()); + const sleep = vi.fn().mockResolvedValue(undefined); + const response = await fetchWithRetry('u', {}, { fetchFn, sleep, attempts: 5 }); + expect(response.status).toBe(200); + expect(fetchFn).toHaveBeenCalledTimes(3); + }); + + it('does not retry a 404', async () => { + const fetchFn = vi.fn().mockResolvedValue(status(404)); + const sleep = vi.fn().mockResolvedValue(undefined); + const response = await fetchWithRetry('u', {}, { fetchFn, sleep, attempts: 5 }); + expect(response.status).toBe(404); + expect(fetchFn).toHaveBeenCalledTimes(1); + expect(sleep).not.toHaveBeenCalled(); + }); + + it('returns the last 429 once attempts are exhausted', async () => { + const fetchFn = vi.fn().mockResolvedValue(status(429)); + const sleep = vi.fn().mockResolvedValue(undefined); + const response = await fetchWithRetry('u', {}, { fetchFn, sleep, attempts: 3 }); + expect(response.status).toBe(429); + expect(fetchFn).toHaveBeenCalledTimes(3); + expect(sleep).toHaveBeenCalledTimes(2); + }); +}); + +describe('retryDelayMs', () => { + it('honours Retry-After in seconds, capped', () => { + expect(retryDelayMs(status(429, { 'Retry-After': '7' }), 0, 60_000)).toBe(7000); + expect(retryDelayMs(status(429, { 'Retry-After': '999' }), 0, 60_000)).toBe(60_000); + }); + + it('backs off exponentially without a header', () => { + const first = retryDelayMs(status(429), 0, 60_000); + const third = retryDelayMs(status(429), 2, 60_000); + expect(third).toBeGreaterThan(first); + }); +});