From 33536d8347611f7e2c8251b896a26f13e163b501 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Tue, 29 Sep 2026 23:05:14 -0400 Subject: [PATCH 01/14] OpenConceptLab/ocl_issues#2849 | Auto Match waits out a busy server instead of freezing or failing rows A 429 now means "slower", never "broken": - services/capacity.js: requestWithCapacityRetry waits for Retry-After plus 0-50% jitter, however often a 429 comes, until a 30-minute cap; then the call ends "throttled". 502/503/504 and network errors are retried twice, then reported as errors. Stop ends any wait. Never rejects. - One gate per server per tab: a 429 on any request holds back the others, and the bulk scheduler (runWithConcurrency) sends no new batch while it's paused instead of draining the queue into refusals. - At most 2 $rerank calls in flight per tab (a limiter), the per-row reranks after each finished batch included. Moved off APIService.post/get (which never settled on a 429 and resolved a 5xx as its body): $rerank, ScispaCy, the bridge $match (through the service oclmap hands the Bridge Match component, so no component release is needed), row-panel search and facets, and the logs POSTs. Bulk and single-row $match, lookups, $resolveReference and the AutomatchRun create/close calls now wait out 429s too. These calls pass handlesThrottle, so the full-screen "Too many requests" countdown, which covered Stop, stays closed. Rows: - A waiting row shows "Waiting for capacity: results may take longer due to demand", in its panel and in the toolbar. - A row still refused at the cap is throttled (-4): "not run, retry", never failed. The run summary lists those rows apart from failures, and a run with any is partial, not failed. - A 5xx on $rerank marks the rerank failed; before, the row looked reranked with no scores. - A stopped or failed run still autosaves and closes its AutomatchRun; the close PATCH retries and ignores Stop. - The logs POST sends one at a time per project, reading the log when it sends, so a retried older POST can't overwrite a newer one. - A bridge call the component never sends (bridging unavailable) marks the algorithm n/a instead of leaving the row running forever. The X-OCL-Capacity-* headers are parsed and logged; adapting to them is OpenConceptLab/ocl_online#340. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/Candidates.jsx | 42 +- src/components/map-projects/MapProject.jsx | 787 +++++++++++++----- .../__tests__/autoMatchRows.test.js | 43 +- .../map-projects/__tests__/autosave.test.js | 62 +- .../map-projects/__tests__/matchBatch.test.js | 211 ++++- .../__tests__/rowProgress.test.js | 59 ++ src/components/map-projects/autoMatchRows.js | 15 + src/components/map-projects/autosave.js | 32 + src/components/map-projects/constants.jsx | 2 + src/components/map-projects/matchBatch.js | 184 ++-- src/components/map-projects/rowProgress.js | 47 ++ src/i18n/locales/en/translations.json | 7 + src/i18n/locales/es/translations.json | 7 + src/i18n/locales/zh/translations.json | 7 + src/services/APIService.js | 8 +- src/services/__tests__/attribution.test.js | 19 + src/services/__tests__/capacity.test.js | 485 +++++++++++ src/services/attribution.js | 16 +- src/services/capacity.js | 314 +++++++ 19 files changed, 2014 insertions(+), 333 deletions(-) create mode 100644 src/components/map-projects/__tests__/rowProgress.test.js create mode 100644 src/components/map-projects/rowProgress.js create mode 100644 src/services/__tests__/capacity.test.js create mode 100644 src/services/capacity.js diff --git a/src/components/map-projects/Candidates.jsx b/src/components/map-projects/Candidates.jsx index 696ce03a..cb2320e1 100644 --- a/src/components/map-projects/Candidates.jsx +++ b/src/components/map-projects/Candidates.jsx @@ -23,6 +23,7 @@ import RefreshIcon from '@mui/icons-material/Refresh'; import SortIcon from '@mui/icons-material/Sort'; import GroupIcon from '@mui/icons-material/Layers'; import AssistantIcon from '@mui/icons-material/Assistant'; +import HourglassIcon from '@mui/icons-material/HourglassEmpty'; import isEmpty from 'lodash/isEmpty' import flatten from 'lodash/flatten' @@ -53,38 +54,7 @@ import { conceptForMapping, getAIAnalysisCandidateIDs } from './viewBuilders.js' - -const getRowProgressLabel = (stageMap, algos) => { - if(stageMap === undefined) - return {label: false} - if(!stageMap) - return {label: 'Preparing...', status: 'partial'} - - const stages = algos.map(k => stageMap[k.id]); - - if (!stages.length || stages.every(v => v === -1)) { - return { label: 'Not started', status: 'idle' }; - } - - const runningIndex = stages.findIndex(v => v === 0); - if (runningIndex !== -1) { - return { - label: `Running: ${algos[runningIndex].id}...`, - status: 'running', - }; - } - - if (stages.every(v => v === 1)) { - return true - } - - const waitingIndex = stages.findIndex(v => v === -1) - if(waitingIndex !== -1) { - return {label: `Waiting: ${algos[waitingIndex].id}`, status: 'waiting'} // AutoMatch Bulk - } - - return { label: 'Partially completed', status: 'partial' }; -} +import { getRowProgressLabel } from './rowProgress.js' const Sort = ({ selected, onSort }) => { const { t } = useTranslation(); @@ -407,7 +377,7 @@ const CandidateList = ({rowViews, header, rowIndex, sortBy, order, openConceptPa // conceptCache — project-wide ConceptDefinition store, keyed by concept_key. // algosSelected — algorithm definitions (for headers/grouping). // (plans/unified-mapper-model.md "How the views map onto this model".) -const Candidates = ({rowIndex, rowState, conceptCache, targetCanonical, targetRelativeUrl, openConceptPanel, showItem, isSelectedForMap, onMap, onFetchMore, isLoading, candidatesScore, repoVersion, analysis, onFetchRecommendation, appliedFacets, setAppliedFacets, filters, facets, columns, defaultFilters, locales, models, selectedModel, onModelChange, promptTemplates, promptTemplate, onPromptTemplateChange, onRefreshClick, rowStage, inAIAssistantGroup, algosSelected, canSelectAIModel}) => { +const Candidates = ({rowIndex, rowState, conceptCache, targetCanonical, targetRelativeUrl, openConceptPanel, showItem, isSelectedForMap, onMap, onFetchMore, isLoading, candidatesScore, repoVersion, analysis, onFetchRecommendation, appliedFacets, setAppliedFacets, filters, facets, columns, defaultFilters, locales, models, selectedModel, onModelChange, promptTemplates, promptTemplate, onPromptTemplateChange, onRefreshClick, rowStage, inAIAssistantGroup, algosSelected, canSelectAIModel, capacityWait=false}) => { const { t } = useTranslation(); const [sortBy, setSortBy] = React.useState('rerank_score') const [groupBy, setGroupBy] = React.useState('quality') @@ -466,7 +436,9 @@ const Candidates = ({rowIndex, rowState, conceptCache, targetCanonical, targetRe const canFetchMore = hasAnyView const algoStagesValue = values(rowStage || {}).filter((_, i) => Object.keys(rowStage || {})[i] !== 'recommend') const areAlgoRun = algoStagesValue.length > 0 && algoStagesValue.every(v => v === 1) - const { label } = getRowProgressLabel(rowStage, algosSelected); + const { label, status: progressStatus } = getRowProgressLabel(rowStage, algosSelected, {t, capacityWait}); + // Waiting out a busy server, or given up on for now: not a spinner (ocl_issues#2849). + const isCapacityStatus = ['capacity_wait', 'throttled'].includes(progressStatus) const byAlgoScore = sortBy === 'algo_score' const byRerankScore = sortBy === 'rerank_score' @@ -628,7 +600,7 @@ const Candidates = ({rowIndex, rowState, conceptCache, targetCanonical, targetRe { !areAlgoRun && label && - } variant='outlined' color='warning' size='small' label={label} sx={{margin: '0 8px'}} /> + : } variant='outlined' color='warning' size='small' label={label} sx={{margin: '0 8px'}} /> } { // Refresh stays visible even when there are zero candidates so diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index 173abdfc..3ba62471 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -71,6 +71,7 @@ import { OperationsContext } from '../app/LayoutContext'; import APIService, { isTransientNetworkError, retryWithBackoff } from '../../services/APIService'; import { buildAttributionHeaders, buildConfigSnapshot, summarizeRunCompletion } from '../../services/attribution' +import { CAPACITY_WAIT_CAP_MS, createCapacityGate, createLimiter, isRetryableError, requestWithCapacityRetry, sleepUnlessCancelled } from '../../services/capacity' import { highlightTexts, dropVersion, getCurrentUser, hasAuthGroup, hasCapability, getMapperPreview, getNewProjectBlockReason, downloadObject, currentUserToken, refreshCurrentUserCapabilitiesCache } from '../../common/utils'; import { WHITE, SURFACE_COLORS, TEXT_GRAY } from '../../common/colors'; @@ -86,8 +87,8 @@ import MapProjectDeleteConfirmDialog from './MapProjectDeleteConfirmDialog'; import ConfigurationForm from './ConfigurationForm' import Controls from './Controls' import DataGridControls from './DataGridControls' -import { getPreviewEligibleRowIndexes, getRowsToProcess, spendsMatchQuota, getRowCapByMatchOperations, shouldStopAIStep, getAIRequestIdempotencyKey, getCandidatePoolFingerprint, hasCurrentAnalysis, getScispacyRowResults, getPendingRowLookups, waitForLookups, AI_LOOKUP_WAIT_MS, RERANK_LOOKUP_WAIT_MS } from './autoMatchRows' -import { createAutosaveScheduler, saveOnLeave, trackSave, whenSaved, installUnloadGuard } from './autosave' +import { getPreviewEligibleRowIndexes, getRowsToProcess, spendsMatchQuota, getRowCapByMatchOperations, shouldStopAIStep, getAIRequestIdempotencyKey, getCandidatePoolFingerprint, hasCurrentAnalysis, getScispacyRowResults, getPendingRowLookups, waitForLookups, AI_LOOKUP_WAIT_MS, RERANK_LOOKUP_WAIT_MS, RERANK_MAX_IN_FLIGHT, isScispacyWarmingUp } from './autoMatchRows' +import { createAutosaveScheduler, createLatestSender, saveOnLeave, trackSave, whenSaved, installUnloadGuard } from './autosave' import MatchSummaryCard from './MatchSummaryCard' import MappingDecisionResult from './MappingDecisionResult' import DecisionSelector from './DecisionSelector' @@ -108,7 +109,7 @@ import QuotaDialog from '../common/QuotaDialog' import { getQuotaError } from '../common/quotaErrors' import MapperQuotaChip from './MapperQuotaChip' import { getPreviewLimitError, isAIPreviewLimitError } from './previewLimits' -import { formatRowNumbers, runMatchBatch, requestSingleMatch } from './matchBatch' +import { formatRowNumbers, runMatchBatch, requestSingleMatch, runWithConcurrency, INTERACTIVE_WAIT_CAP_MS } from './matchBatch' import { getAIAssistantChoices, getModelUsed, getPromptTemplateKey, createLatestRequestGate } from './aiVisibility' import { DEFAULT_ENCODER_MODEL } from './rerankerModels' import { normalizeAlgorithmInvocation, hasSuccessfulAlgorithmResponse, getAlgorithmStagesFromResponses, lookupStatusRank, buildRecommendableConceptEntry, stripConstantClassAndDatatype, buildLookupConceptUrl, serializeCandidates, toStoredRowMatchState } from './normalizers' @@ -226,6 +227,28 @@ const MapProject = () => { const autoMatchRunGateRef = React.useRef(null) if(!autoMatchRunGateRef.current) autoMatchRunGateRef.current = createLatestRequestGate() + // The latest run's ticket, for waits that must end when that run stops. + const autoMatchRunTicketRef = React.useRef(null) + // Waiting out a busy server (ocl_issues#2849). One gate per server: a 429 on + // any request holds back this tab's others to that server, and the Auto + // Match scheduler. At most RERANK_MAX_IN_FLIGHT $rerank calls at once, the + // per-row reranks during a run included. + const capacityGatesRef = React.useRef(new Map()) + const rerankLimiterRef = React.useRef(null) + if(!rerankLimiterRef.current) + rerankLimiterRef.current = createLimiter(RERANK_MAX_IN_FLIGHT) + // The requests waiting for capacity now, each with its rows, for the + // "Waiting for capacity" notices. + const capacityWaitsRef = React.useRef(new Map()) + const [capacityWaitRows, setCapacityWaitRows] = React.useState(null) + // A run's rows the server stayed too busy for, by algorithm id ('rerank' + // included), for the end-of-run notice. + const throttledRunRowsRef = React.useRef({}) + // One logs POST at a time per project URL (see createLatestSender). + const logsSendersRef = React.useRef(new Map()) + // The row a bridge $match is for, set around the call into the Bridge Match + // component (see getBridgeMatchService). + const bridgeCallContextRef = React.useRef(null) // ai_assistant.calls is a one-time allowance (no reset, R2) - once exhausted it // stays exhausted for the rest of the session, so this is never reset per-run. const aiQuotaExhaustedRef = React.useRef(false); @@ -1555,7 +1578,6 @@ const MapProject = () => { }) return } - const rowLogsForSave = options.logs || logsRef.current const projectLogsForSave = options.projectLogs || projectLogsRef.current setIsSaving(true) const f = getFileObjectFromRows() @@ -1654,7 +1676,7 @@ const MapProject = () => { if(!isAutoSave) baseSetAlert({severity: 'success', message: t('map_project.successfully_saved'), duration: 2000}) - APIService.new().overrideURL(response.data.url).appendToUrl('logs/').post({logs: {row_logs: rowLogsForSave, project_logs: savedProjectLogs}}).then(() => {}) + postProjectLogs(response.data.url) } else if(options.leaving && status !== 401) { // The user has left the project, so no dialog can show: say so once. baseSetAlert({severity: 'error', message: t('map_project.leave_save_failed', {name: name || project?.name, detail: errorData?.detail || t('unknown_error')}), duration: 12000}) @@ -1690,11 +1712,98 @@ const MapProject = () => { projectLogsRef.current = newLogs setProjectLogs(newLogs) if(project?.url) - APIService.new().overrideURL(project.url).appendToUrl('logs/').post({logs: {row_logs: logsRef.current, project_logs: newLogs}}).then(() => {}) + postProjectLogs(project.url) + } + + // POSTs the project's logs as they stand when it sends, waiting out a busy + // server. One POST at a time per project: each sends the whole log, so an + // older one that waited out a 429 mustn't land after a newer one + // (ocl_issues#2849). Best effort, like before. + const postProjectLogs = url => { + if(!url) + return + if(!logsSendersRef.current.has(url)) + logsSendersRef.current.set(url, createLatestSender(() => requestWithCapacityRetry( + () => APIService.new().overrideURL(url).appendToUrl('logs/').request( + 'POST', + {logs: {row_logs: logsRef.current, project_logs: projectLogsRef.current}}, + null, + {handlesThrottle: true} + ), + {maxWaitMs: INTERACTIVE_WAIT_CAP_MS} + ))) + logsSendersRef.current.get(url).trigger() } const scheduleAutoSave = reason => autosaveSchedulerRef.current.schedule(reason) + // ── Waiting out a busy server (ocl_issues#2849) ─────────────────────────── + // The gate for a request's server, shared by every request this tab sends it. + const getCapacityGate = url => { + let origin + try { + origin = new URL(url, window.location.href).origin + } catch (_) { + origin = String(url) + } + if(!capacityGatesRef.current.has(origin)) + capacityGatesRef.current.set(origin, createCapacityGate()) + return capacityGatesRef.current.get(origin) + } + + const syncCapacityWaits = () => { + const rows = {} + capacityWaitsRef.current.forEach(rowIndexes => rowIndexes.forEach(index => { rows[index] = true })) + setCapacityWaitRows(capacityWaitsRef.current.size ? rows : null) + } + + // onWait/onWaitEnd for one request on these rows. While it waits out a busy + // server (a 429, or a pause another request's 429 started; not an error + // backoff), its rows show "Waiting for capacity", and its first wait goes in + // each row's log with the server's capacity headers. + const trackCapacityWait = (rowIndexes, logExtras = {}) => { + const waitId = {} + let logged = false + return { + onWait: ({reason, retryAfterMs, capacity}) => { + if(reason === 'error') + return + capacityWaitsRef.current.set(waitId, rowIndexes) + syncCapacityWaits() + if(logged) + return + logged = true + rowIndexes.forEach(index => log({ + action: 'capacity_wait', + description: t('map_project.waiting_for_capacity'), + extras: {...logExtras, reason, retry_after_ms: retryAfterMs ?? null, ...(capacity ? {capacity} : {})} + }, index)) + }, + onWaitEnd: () => { + if(capacityWaitsRef.current.delete(waitId)) + syncCapacityWaits() + }, + } + } + + // The server's X-OCL-Capacity-* headers are only logged for now; adapting to + // them is OpenConceptLab/ocl_online#340. + const noteCapacity = capacity => console.debug('OCL capacity', capacity) + + // Stop for the run in progress: true once the user stops it, or a newer run + // starts. Waits in a run poll it. + const getRunStopCheck = () => { + const ticket = autoMatchRunTicketRef.current + return () => Boolean(abortRef.current || ticket !== autoMatchRunTicketRef.current) + } + + const noteThrottledRunRow = (algoId, index) => { + throttledRunRowsRef.current = { + ...throttledRunRowsRef.current, + [algoId]: uniq([...(throttledRunRowsRef.current[algoId] || []), index]), + } + } + // ── ocl_online#105 Phase 5: AutomatchRun attribution ────────────────────── // Build the X-OCL-Request-Source + X-OCL-Event-Metadata headers for a single // per-row backend call. Attribution is by explicit ORIGIN, not ambient: only @@ -1737,10 +1846,16 @@ const MapProject = () => { }), } try { - const response = await APIService.new().overrideURL(project.url).appendToUrl('auto-match-runs/') - .post(body, null, {}, undefined, true) - const status = response?.response?.status || response?.status - const data = response?.response?.data || response?.data + // A busy server's 429 is waited out, and Stop ends the wait. Nothing + // else is retried: after a 502 the run may exist already, and a retry + // would open a second one (ocl_issues#2849). + const result = await requestWithCapacityRetry( + () => APIService.new().overrideURL(project.url).appendToUrl('auto-match-runs/').request('POST', body, null, {handlesThrottle: true}), + {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, isCancelled: getRunStopCheck(), isRetryable: () => false, onCapacity: noteCapacity} + ) + const response = result.ok ? result.response : result.error?.response + const status = response?.status + const data = response?.data const id = data?.id if(id) { automatchRunRef.current = {id, algoIds: map(selectedAlgos, 'id')} @@ -1753,14 +1868,14 @@ const MapProject = () => { setPreviewLimit({errorCode: data.error_code, limit: data.limit, used: data.used}) return } - projectLog({action: 'automatch_run_create_failed', extras: {status: status || 'no-id'}}) + projectLog({action: 'automatch_run_create_failed', extras: {status: status || 'no-id', ...(result.ok ? {} : {reason: result.reason, error: result.error?.message || null})}}) } catch (err) { projectLog({action: 'automatch_run_create_failed', extras: {error: err?.message || 'unknown'}}) } } // PATCH the run to a terminal status at run end. completed/failed are derived - // from the per-row algo stages (rowStageRef: -2 failed, 1 done, -1 not run) + // from the per-row algo stages (rowStageRef: -4 throttled, -2 failed, 1 done, -1 not run) // over the rows we set out to process; a row counts as failed only when EVERY // attempted algo failed for it. Never throws — a failed PATCH (or a run that // was never created) must not surface to the user. @@ -1775,10 +1890,13 @@ const MapProject = () => { aborted: abortRef.current, stoppedForQuota: Boolean(matchQuotaStopRef.current), }) - try { - await APIService.new().overrideURL('/auto-match-runs/' + run.id + '/') - .request('PATCH', payload) - } catch (_) { /* attribution is best-effort; never block on the PATCH */ } + // Closing the run is idempotent, so a busy server or a gateway error is + // retried; Stop doesn't end it, so a stopped run is closed too + // (ocl_issues#2849). requestWithCapacityRetry never rejects. + await requestWithCapacityRetry( + () => APIService.new().overrideURL('/auto-match-runs/' + run.id + '/').request('PATCH', payload, null, {handlesThrottle: true}), + {maxWaitMs: INTERACTIVE_WAIT_CAP_MS} + ) } const fetchRepo = (url, _repo) => APIService.new().overrideURL(url).get().then(response => setRepo(response.data?.id ? response.data : _repo)) @@ -1923,6 +2041,42 @@ const MapProject = () => { return service } + // The service handed to the Bridge Match component (react-bridge-match), + // which sends the bridge $match through its post(). APIService.post never + // settled on a 429, which froze a run's bridge step, and resolved a 5xx as + // its body. This post() waits out 429s, retries transient failures, and + // always settles with what the component reads: the response, or a body + // with a detail (ocl_issues#2849). A new instance each render, as before: + // the component points it at its own URL for a GET. + const getBridgeMatchService = () => { + const service = getMatchAPIService() + service.post = (body, token, headers, query) => { + const context = bridgeCallContextRef.current || {} + context.posted = true + return requestWithCapacityRetry(() => service.request('POST', body, token, {headers: headers || {}, query, handlesThrottle: true}), { + gate: getCapacityGate(service.URL), + isCancelled: context.isCancelled || (() => false), + maxWaitMs: context.isBulk ? CAPACITY_WAIT_CAP_MS : INTERACTIVE_WAIT_CAP_MS, + onCapacity: noteCapacity, + ...(isNumber(context.rowIndex) ? trackCapacityWait([context.rowIndex], getAlgoLogExtras(bridgeAlgo)) : {}), + }).then(result => { + if(result.ok) + return result.response + if(result.reason === 'throttled') + return {detail: t('map_project.algorithm_throttled'), status: 429, throttled: true} + if(result.reason === 'cancelled') + return {detail: 'cancelled', cancelled: true} + const data = result.error?.response?.data + const hasServerBody = data && typeof data === 'object' && !Array.isArray(data) && (data.detail || data.error_code) + return hasServerBody ? data : { + detail: t('map_project.match_request_failed', {error: result.error?.response?.data?.detail || result.error?.message || t('unknown_error')}), + status: result.error?.response?.status || null, + } + }) + } + return service + } + // Pick the highest-scoring target-repo ConceptRow for a given rowIndex // from the unified-model state. Returns a RowView ({candidate, // conceptDefinition, conceptRow, bridgeConceptDefinition?}) or null. @@ -2057,7 +2211,9 @@ const MapProject = () => { // reported at the end of the run so the user knows which rows to re-run. const failedMatchRows = {} const runTicket = autoMatchRunGateRef.current.next() + autoMatchRunTicketRef.current = runTicket const isCurrentRun = () => autoMatchRunGateRef.current.isCurrent(runTicket) + throttledRunRowsRef.current = {} // Function to process a single batch const processBatch = async (_repo, rowBatch, algo) => { @@ -2091,7 +2247,8 @@ const MapProject = () => { const logExtras = getAlgoLogExtras(algo) // service.request rejects on a network error or a non-2xx; service.post // would resolve, and the failed batch would pass as a success with no - // candidates (ocl_online#257). runMatchBatch retries transient failures. + // candidates (ocl_online#257). runMatchBatch retries transient failures, + // and waits out a busy server's 429s (ocl_issues#2849). return runMatchBatch({ rowIndexes, send: attempt => service.request( @@ -2104,9 +2261,12 @@ const MapProject = () => { includeSearchMeta: true, ...(algo.query_params || {}), ...extraParams - } + }, + handlesThrottle: true, } ), + retryOptions: {gate: getCapacityGate(service.URL), onCapacity: noteCapacity}, + ...trackCapacityWait(rowIndexes, logExtras), // A batch that outlives its run must not touch the next run's stages. setStage: (index, stage) => { if(isCurrentRun()) @@ -2118,6 +2278,12 @@ const MapProject = () => { if(!previewLimit) failedMatchRows[algo.id] = [...(failedMatchRows[algo.id] || []), index] }, + // Still refused after the long cap: not run, and not a failure. + onRowThrottled: (index, {attempts, waitedMs}) => { + log({action: 'algo_throttled', description: t('map_project.row_throttled'), extras: {...logExtras, attempts, waited_ms: waitedMs}}, index) + if(isCurrentRun()) + noteThrottledRunRow(algo.id, index) + }, // Sets matchQuotaStopRef, which stops the rest of the $match requests // but lets the run finish with what it has. onPreviewLimit: err => handlePreviewLimitError(err), @@ -2133,66 +2299,49 @@ const MapProject = () => { const processWithConcurrency = async (_repo, algo, _rows) => { // A project's saved settings run within this user's limits. const { batchSize, concurrentRequests } = getRequestSettings(algo, fullRequestLimits) - const CHUNK_SIZE = batchSize // Number of rows per batch - const MAX_CONCURRENT_REQUESTS = concurrentRequests; // Number of parallel API requests allowed - const rowChunks = chunk(_rows, CHUNK_SIZE); - - const queue = rowChunks.slice(); // Copy of all chunks to be processed - const activeRequests = new Set(); - - while (queue.length > 0 || activeRequests.size > 0) { - // Fill activeRequests up to MAX_CONCURRENT_REQUESTS - while (queue.length > 0 && activeRequests.size < MAX_CONCURRENT_REQUESTS) { - if (abortRef.current) { - setLoadingMatches(false) - return - }; - if (matchQuotaStopRef.current && spendsMatchQuota(algo, {canBridge})) { - queue.length = 0 - break - } - const rowBatch = queue.shift(); - const promise = processBatch(_repo, rowBatch, algo).then((data) => { - // Populate rowMatchState before any consumer (setStateViews / - // setAutoMatched) tries to read it. The per-row fetch flow - // routes through `onResponse` which already calls - // mergeIntoRowMatchState; bulk processBatch doesn't, so we - // run the normalizer here. - if(Array.isArray(data) && data.length) { - const projectCtx = buildProjectContext() - if(projectCtx) { - const algoCfg = ensureConceptIdentity(algo) - if(algoCfg) { - data.forEach(rowEntry => { - const idx = rowEntry?.row?.__index - if(!isNumber(idx)) return - mergeIntoRowMatchState(idx, normalizeAlgorithmInvocation(rowEntry, { - algorithmId: algo.id, - algorithmConfig: algoCfg, - projectContext: projectCtx, - rowIndex: idx, - // Bulk auto-match's $match request sends reranker=!isMultiAlgo - // (line 1433). Trust the server's normalized score only when - // that flag was true — otherwise it's a per-algo native score - // (e.g. FAISS similarity × 100) masquerading as a rerank. - trustServerRerank: !isMultiAlgo - }), {isRunTraffic: true}) - }) - } + // While a 429 holds the server's gate, no new batch goes out: the queue + // waits instead of draining into refusals (ocl_issues#2849). + const finished = await runWithConcurrency(chunk(_rows, batchSize), { + concurrency: concurrentRequests, + gate: getCapacityGate(getMatchAPIService(algo).URL), + shouldAbort: () => abortRef.current, + shouldSkipRest: () => Boolean(matchQuotaStopRef.current && spendsMatchQuota(algo, {canBridge})), + run: rowBatch => processBatch(_repo, rowBatch, algo).then((data) => { + // Populate rowMatchState before any consumer (setStateViews / + // setAutoMatched) tries to read it. The per-row fetch flow + // routes through `onResponse` which already calls + // mergeIntoRowMatchState; bulk processBatch doesn't, so we + // run the normalizer here. + if(Array.isArray(data) && data.length) { + const projectCtx = buildProjectContext() + if(projectCtx) { + const algoCfg = ensureConceptIdentity(algo) + if(algoCfg) { + data.forEach(rowEntry => { + const idx = rowEntry?.row?.__index + if(!isNumber(idx)) return + mergeIntoRowMatchState(idx, normalizeAlgorithmInvocation(rowEntry, { + algorithmId: algo.id, + algorithmConfig: algoCfg, + projectContext: projectCtx, + rowIndex: idx, + // Bulk auto-match's $match request sends reranker=!isMultiAlgo + // (line 1433). Trust the server's normalized score only when + // that flag was true — otherwise it's a per-algo native score + // (e.g. FAISS similarity × 100) masquerading as a rerank. + trustServerRerank: !isMultiAlgo + }), {isRunTraffic: true}) + }) } } - if(!isMultiAlgo) - setStateViews(data, _repo) - setMatchedConcepts(prev => [...prev, ...data]); - activeRequests.delete(promise); // Remove from active set after completion - }); - activeRequests.add(promise); - } - - // Wait for at least one request to complete before continuing - if (activeRequests.size > 0) - await Promise.race(activeRequests); - } + } + if(!isMultiAlgo) + setStateViews(data, _repo) + setMatchedConcepts(prev => [...prev, ...data]); + }), + }) + if(!finished) + setLoadingMatches(false) }; const processRerankWithConcurrency = async (_rows, maxConcurrent = 2) => { @@ -2283,6 +2432,9 @@ const MapProject = () => { // closed out (completed / partial / failed / cancelled) via the finally, // even on cancel or an unexpected throw mid-pipeline. await createAutomatchRun(_selectedAlgos, rowsToProcess) + // False when the run was refused (a preview limit) or stopped before it + // began: it changed nothing to save. + const runStarted = !abortRef.current try { bulkMatchAlgoIdsRef.current = map(_selectedAlgos, 'id') // Reset all algo stages to -1 for every row before starting so that @@ -2330,26 +2482,43 @@ const MapProject = () => { setLoadingMatches(false) setEndMatchingAt(moment()) } + const getAlgoLabel = algoId => { + if(algoId === 'rerank') + return t('map_project.rerank_step') + const name = find(_selectedAlgos, {id: algoId})?.name + return isString(name) && name ? name : algoId + } + const getRowsByAlgo = rowsPerAlgo => { + const rowsByAlgo = {} + keys(rowsPerAlgo).forEach(algoId => { rowsByAlgo[algoId] = uniq(rowsPerAlgo[algoId]).sort((a, b) => a - b) }) + return rowsByAlgo + } + const describeRows = rowsByAlgo => keys(rowsByAlgo).map(algoId => t('map_project.auto_match_rows_failed_algorithm', { + algorithm: getAlgoLabel(algoId), + rows: formatRowNumbers(rowsByAlgo[algoId]), + })).join('; ') + const endOfRunNotices = [] const failedAlgoIds = keys(failedMatchRows) if(!abortRef.current && failedAlgoIds.length) { - const rowsByAlgo = {} - failedAlgoIds.forEach(algoId => { rowsByAlgo[algoId] = uniq(failedMatchRows[algoId]).sort((a, b) => a - b) }) - const getAlgoLabel = algoId => { - const name = find(_selectedAlgos, {id: algoId})?.name - return isString(name) && name ? name : algoId - } + const rowsByAlgo = getRowsByAlgo(failedMatchRows) projectLog({action: 'auto_match_rows_failed', extras: {row_indexes_by_algorithm: rowsByAlgo}}) - setAlert({ - severity: 'warning', - message: t('map_project.auto_match_rows_failed', { - count: uniq(flatten(values(rowsByAlgo))).length, - details: failedAlgoIds.map(algoId => t('map_project.auto_match_rows_failed_algorithm', { - algorithm: getAlgoLabel(algoId), - rows: formatRowNumbers(rowsByAlgo[algoId]), - })).join('; '), - }), - }) + endOfRunNotices.push(t('map_project.auto_match_rows_failed', { + count: uniq(flatten(values(rowsByAlgo))).length, + details: describeRows(rowsByAlgo), + })) } + // Rows the server stayed too busy for weren't run: say so apart from + // the failures (ocl_issues#2849). + if(!abortRef.current && keys(throttledRunRowsRef.current).length) { + const rowsByAlgo = getRowsByAlgo(throttledRunRowsRef.current) + projectLog({action: 'auto_match_rows_throttled', extras: {row_indexes_by_algorithm: rowsByAlgo}}) + endOfRunNotices.push(t('map_project.auto_match_rows_throttled', { + count: uniq(flatten(values(rowsByAlgo))).length, + details: describeRows(rowsByAlgo), + })) + } + if(endOfRunNotices.length) + setAlert({severity: 'warning', message: endOfRunNotices.join(' ')}) if(!abortRef.current && matchQuotaStopRef.current) projectLog({action: 'auto_match_stopped_for_quota', extras: {reason: matchQuotaStopRef.current}}) if(!abortRef.current) @@ -2367,9 +2536,11 @@ const MapProject = () => { } : {}) } }) - if(!abortRef.current) - scheduleAutoSave('auto_match') } finally { + // Save what the run matched, also when it was stopped or failed part + // way (ocl_issues#2849). + if(runStarted) + scheduleAutoSave(abortRef.current ? 'auto_match_stopped' : 'auto_match') await completeAutomatchRun(_selectedAlgos, rowsToProcess) refreshMapperQuotaCache() } @@ -3531,7 +3702,8 @@ const MapProject = () => { // resolved, so a failed call showed as "no candidates" (ocl_online#283). // requestSingleMatch retries it like a bulk batch; a call that still fails // reaches the callback as an error body, which shows the error and marks - // the algorithm failed for the row. + // the algorithm failed for the row. A 429 is waited out, and a server + // still busy at the cap marks the algorithm throttled (ocl_issues#2849). requestSingleMatch(() => service.request( 'POST', payload, @@ -3549,11 +3721,18 @@ const MapProject = () => { semantic: ['ocl-semantic', 'custom'].includes(algoDef.type), reranker: !isMultiAlgo && algoDef.provider === 'ocl', encoder_model: !isMultiAlgo && encoderModel ? encoderModel : undefined - } + }, + handlesThrottle: true, } - )).then(({ok, response, errorBody, previewLimit}) => { + ), {retryOptions: { + gate: getCapacityGate(service.URL), + onCapacity: noteCapacity, + ...trackCapacityWait([__row.__index], getAlgoLogExtras(algoDef)), + }}).then(({ok, response, errorBody, previewLimit, throttled}) => { if(ok) return callback(response, payload) + if(throttled) + return callback({detail: t('map_project.algorithm_throttled'), status: 429, throttled: true}, payload) // A preview limit keeps the server's body, which opens the limit dialog. callback(previewLimit ? errorBody : {...errorBody, detail: t('map_project.match_request_failed', {error: errorBody.detail})}, payload) }) @@ -3626,13 +3805,18 @@ const MapProject = () => { return clearRefreshRowStageSnapshot(__row.__index) refreshMapperQuotaCache() - log({action: 'algo_failed', extras: {...logExtras, error: response.detail, status: response.status, ...(offset ? {offset} : {})}}, __row.__index) - setAlert({message: response.detail, severity: 'error'}) + // The server stayed too busy: the algorithm didn't run, it didn't + // fail, and no failed response is kept (ocl_issues#2849). + const isThrottled = Boolean(response.throttled) + log({action: isThrottled ? 'algo_throttled' : 'algo_failed', ...(isThrottled ? {description: t('map_project.row_throttled')} : {}), extras: {...logExtras, error: response.detail, status: response.status, ...(offset ? {offset} : {})}}, __row.__index) + setAlert({message: response.detail, severity: isThrottled ? 'warning' : 'error'}) if(offset) { // A failed "load more" keeps the pages already loaded. The // algorithm stays done if it loaded a page; "load more" also runs // algorithms that never did, and those stay failed. - markAlgo(__row.__index, algoId, hasSuccessfulAlgorithmResponse(rowMatchStateRef.current?.[__row.__index], algoId) ? 1 : -2) + markAlgo(__row.__index, algoId, hasSuccessfulAlgorithmResponse(rowMatchStateRef.current?.[__row.__index], algoId) ? 1 : (isThrottled ? -4 : -2)) + } else if(isThrottled) { + markAlgo(__row.__index, algoId, -4) } else { markAlgo(__row.__index, algoId, -2) mergeIntoRowMatchState(__row.__index, normalizeAlgorithmInvocation(null, { @@ -3780,73 +3964,96 @@ const MapProject = () => { // the next row visit retries. const SCISPACY_WARMUP_RETRY_MS = 2 * 60 * 1000 const SCISPACY_WARMUP_MAX_MS = 10 * 60 * 1000 + const SCISPACY_ALGO_ID = 'ocl-scispacy-loinc' + const SCISPACY_ERROR_MESSAGE = "OCL's scispacy matching service is starting up. This may take a couple minutes. You can safely leave this row and come back. Click Refresh if results aren't here in a couple of minutes." const warmupStart = Date.now() - let warmingUp = true let seenWarmingUp = false + // A run's waits end when it stops; a row matched by hand isn't stopped. + // (abortRef stays set after a Stop, so it can't gate a manual call.) + const isCancelled = isBulk ? getRunStopCheck() : () => false + const capacityWait = trackCapacityWait([__row.__index], {algo: SCISPACY_ALGO_ID}) + const endStopped = () => { + markAlgo(__row.__index, SCISPACY_ALGO_ID, -1) + setIsLoadingInDecisionView(false) + } - while(warmingUp) { - if(abortRef.current) { setIsLoadingInDecisionView(false); return } - try { + for(;;) { + // service.request rejects on a non-2xx: service.post resolved a 5xx as + // its body, and never settled on a 429 (ocl_issues#2849). A 429 waits + // for Retry-After; a network error or a gateway error is retried, but + // not a warm-up answer, which waits two minutes below. + const result = await requestWithCapacityRetry(() => service.request('POST', payload, null, { // ocl_online#105 Phase 5: the scispacy service records its own // api_transactions (ocl-scispacy-loinc) via its AnalyticsMiddleware, so // attribute the $match like the others. Per-row (single-row payload) → // scalar row_index; isBulk distinguishes a run call from a manual one. - const response = await service.post(payload, null, attrHeaders({rowIndex: __row.__index, algorithmId: 'ocl-scispacy-loinc', isRunTraffic: isBulk})) - - // APIService resolves 5xx errors with error.response.data, so response - // is the parsed body object — not the axios response. A 503 warming_up - // from the lambda resolves as {status: 'warming_up', message: '...'}. - // A 502 (EC2 still booting) resolves as {error: '...'}. - const isWarmingUp = response?.status === 'warming_up' || (seenWarmingUp && response?.error) - if(isWarmingUp) { - seenWarmingUp = true - const elapsed = Date.now() - warmupStart - if(elapsed + SCISPACY_WARMUP_RETRY_MS > SCISPACY_WARMUP_MAX_MS) { - markAlgo(__row.__index, 'ocl-scispacy-loinc', -2) - log({action: 'algo_failed', extras: {algo: 'ocl-scispacy-loinc', status: 503, detail: 'warming_up_timeout'}}, __row.__index) - setAlert({message: "OCL's scispacy service did not come up within 10 minutes.", severity: 'error'}) - setIsLoadingInDecisionView(false) - return response - } - // Don't cover an error an earlier algorithm for this row showed. - setAlert(prev => (prev?.severity === 'error' ? prev : {message: t('map_project.scispacy_warming_up'), severity: 'info'})) - await new Promise(resolve => setTimeout(resolve, SCISPACY_WARMUP_RETRY_MS)) - } else { - warmingUp = false - const isError = response?.detail - || response?.status >= 400 - || (response && response.data === undefined && response.status !== 200) - if(isError) { - if(handlePreviewLimitError(response, __row.__index, 'ocl-scispacy-loinc')) - return response - clearRefreshRowStageSnapshot(__row.__index) - refreshMapperQuotaCache() - markAlgo(__row.__index, 'ocl-scispacy-loinc', -2) - log({action: 'algo_failed', extras: {algo: 'ocl-scispacy-loinc', status: response?.status, detail: response?.detail}}, __row.__index) - setAlert({ - message: response?.detail || "OCL's scispacy matching service is starting up. This may take a couple minutes. You can safely leave this row and come back. Click Refresh if results aren't here in a couple of minutes.", - severity: 'warning' - }) - setIsLoadingInDecisionView(false) - return response - } - // Clear ScispaCy's own warming-up or starting notice, but not an - // error another algorithm for this row showed (ocl_online#283). - setAlert(prev => (prev?.severity === 'error' ? prev : false)) - if(callback) callback(response, payload) + headers: attrHeaders({rowIndex: __row.__index, algorithmId: SCISPACY_ALGO_ID, isRunTraffic: isBulk}), + handlesThrottle: true, + }), { + gate: getCapacityGate(service.URL), + isCancelled, + maxWaitMs: isBulk ? CAPACITY_WAIT_CAP_MS : INTERACTIVE_WAIT_CAP_MS, + isRetryable: err => !isScispacyWarmingUp(err, seenWarmingUp) && isRetryableError(err), + onCapacity: noteCapacity, + ...capacityWait, + }) + + if(result.ok) { + const response = result.response + // Clear ScispaCy's own warming-up or starting notice, but not an + // error another algorithm for this row showed (ocl_online#283). + setAlert(prev => (prev?.severity === 'error' ? prev : false)) + if(callback) callback(response, payload) + setIsLoadingInDecisionView(false) + return response + } + if(result.reason === 'cancelled') { + endStopped() + return + } + if(result.reason === 'throttled') { + // Not run, not failed: the row can be run again. + markAlgo(__row.__index, SCISPACY_ALGO_ID, -4) + log({action: 'algo_throttled', description: t('map_project.row_throttled'), extras: {algo: SCISPACY_ALGO_ID, attempts: result.attempts, waited_ms: result.waitedMs}}, __row.__index) + if(isBulk) + noteThrottledRunRow(SCISPACY_ALGO_ID, __row.__index) + setAlert(prev => (prev?.severity === 'error' ? prev : {message: t('map_project.algorithm_throttled'), severity: 'warning'})) + setIsLoadingInDecisionView(false) + return + } + + const err = result.error + const body = err?.response?.data + // The service's host is booting: it answers 503 warming_up, then 502 + // until the app is up. + if(isScispacyWarmingUp(err, seenWarmingUp)) { + seenWarmingUp = true + const elapsed = Date.now() - warmupStart + if(elapsed + SCISPACY_WARMUP_RETRY_MS > SCISPACY_WARMUP_MAX_MS) { + markAlgo(__row.__index, SCISPACY_ALGO_ID, -2) + log({action: 'algo_failed', extras: {algo: SCISPACY_ALGO_ID, status: 503, detail: 'warming_up_timeout'}}, __row.__index) + setAlert({message: "OCL's scispacy service did not come up within 10 minutes.", severity: 'error'}) setIsLoadingInDecisionView(false) - return response + return body } - } catch(err) { - warmingUp = false - markAlgo(__row.__index, 'ocl-scispacy-loinc', -2) - log({action: 'algo_failed', extras: {algo: 'ocl-scispacy-loinc', error: err?.message}}, __row.__index) - setAlert({ - message: "OCL's scispacy matching service is starting up. This may take a couple minutes. You can safely leave this row and come back. Click Refresh if results aren't here in a couple of minutes.", - severity: 'warning' - }) - setIsLoadingInDecisionView(false) + // Don't cover an error an earlier algorithm for this row showed. + setAlert(prev => (prev?.severity === 'error' ? prev : {message: t('map_project.scispacy_warming_up'), severity: 'info'})) + if(!(await sleepUnlessCancelled(SCISPACY_WARMUP_RETRY_MS, isCancelled))) { + endStopped() + return + } + continue } + + if(handlePreviewLimitError(err, __row.__index, SCISPACY_ALGO_ID)) + return body + clearRefreshRowStageSnapshot(__row.__index) + refreshMapperQuotaCache() + markAlgo(__row.__index, SCISPACY_ALGO_ID, -2) + log({action: 'algo_failed', extras: {algo: SCISPACY_ALGO_ID, status: err?.response?.status || null, detail: body?.detail, error: err?.message}}, __row.__index) + setAlert({message: body?.detail || SCISPACY_ERROR_MESSAGE, severity: 'warning'}) + setIsLoadingInDecisionView(false) + return body } } @@ -3966,14 +4173,53 @@ const MapProject = () => { inFlightRerankRef.current.add(index) markAlgo(index, 'rerank', 0) const service = APIService.concepts().appendToUrl('$rerank/') + // A run's rerank ends when the run stops; a row reranked by hand doesn't. + const isCancelled = isRunTraffic ? getRunStopCheck() : () => false + let release = null try { - const response = await service.post({ + // At most RERANK_MAX_IN_FLIGHT $rerank calls at once, whatever fired + // them: every finished batch used to fire one per row (ocl_issues#2849). + release = await rerankLimiterRef.current.acquire({isCancelled}) + if(!release) { + // Stopped before its turn; a stopped run reranks nothing more. + rerankRerunNeededRef.current.delete(index) + markAlgo(index, 'rerank', -1) + return null + } + // Score the candidates that arrived while this call waited its turn too. + const latestRerankRows = buildRerankRowsForRow(index) + // service.request rejects on a non-2xx; service.post resolved a 5xx as if + // it scored nothing, and never settled on a 429. A 429 is waited out. + const result = await requestWithCapacityRetry(() => service.request('POST', { q: query, - rows: rerankRows, + rows: latestRerankRows.length ? latestRerankRows : rerankRows, ...(encoderModel ? { encoder_model: encoderModel } : {}) - }, null, attrHeaders({rowIndex: index, algorithmId: 'reranker', isRunTraffic})) - if(handlePreviewLimitError(response, index, 'rerank', {isRun: isRunTraffic, stopPhase: 'rerank'})) - return response + }, null, {headers: attrHeaders({rowIndex: index, algorithmId: 'reranker', isRunTraffic}), handlesThrottle: true}), { + gate: getCapacityGate(service.URL), + isCancelled, + onCapacity: noteCapacity, + ...trackCapacityWait([index], {algo: 'reranker'}), + }) + if(result.reason === 'cancelled') { + rerankRerunNeededRef.current.delete(index) + markAlgo(index, 'rerank', -1) + return null + } + if(result.reason === 'throttled') { + log({action: 'rerank_throttled', description: t('map_project.row_throttled'), extras: {attempts: result.attempts, waited_ms: result.waitedMs}}, index) + markAlgo(index, 'rerank', -4) + if(isRunTraffic) + noteThrottledRunRow('rerank', index) + return null + } + if(!result.ok) { + if(handlePreviewLimitError(result.error, index, 'rerank', {isRun: isRunTraffic, stopPhase: 'rerank'})) + return null + log({action: 'rerank_failed', description: `Rerank failed with ${encoderModel}`, extras: {status: result.error?.response?.status || null, error: result.error?.response?.data?.detail || result.error?.message || null}}, index) + markAlgo(index, 'rerank', -2) + return null + } + const response = result.response // Write rerank_score into the row's ConceptRows. matchRerankResultToKey // throws on canonical-identity miss; surface to the alert state so a @@ -4017,12 +4263,11 @@ const MapProject = () => { setTimeout(() => setAutoMatched([index]), 1000) return response } catch (e) { - if(handlePreviewLimitError(e, index, 'rerank', {isRun: isRunTraffic, stopPhase: 'rerank'})) - return null - log({action: 'rerank_failed', description: `Rerank failed with ${encoderModel}`}, index) + log({action: 'rerank_failed', description: `Rerank failed with ${encoderModel}`, extras: {error: e?.message || null}}, index) markAlgo(index, 'rerank', -2) return null } finally { + release?.() inFlightRerankRef.current.delete(index) // If new ConceptRows arrived while we were in flight, fire again. if(rerankRerunNeededRef.current.has(index)) { @@ -4060,15 +4305,19 @@ const MapProject = () => { // to mapper-ui-manual. Internally consistent: if we treat this as a // bulk-run rerank for gating, we attribute it to the run. const isRunTraffic = Boolean(isBulkMatchRunningRef.current && bulkMatchAlgoIdsRef.current.length > 0) + // An algorithm the server stayed too busy for (-4), or that isn't + // available to this user (-3), is settled too: the row's other + // candidates still get ranked (ocl_issues#2849). + const isSettled = value => value === 1 || value === -2 || value === -3 || value === -4 if(isRunTraffic) { // Bulk auto-match: wait until all selected algos are done (success or failed) for this row. - const allDone = bulkMatchAlgoIdsRef.current.every(id => stage[id] === 1 || stage[id] === -2) + const allDone = bulkMatchAlgoIdsRef.current.every(id => isSettled(stage[id])) if(!allDone) return } else { // Single-row: wait until every configured algo has either completed (1) or failed (-2). // Checking only === 0 (in-flight) wasn't enough — algos not yet started (-1) or // uninitialized (undefined) would let rerank fire before their candidates arrived. - if(selectedAlgoIds.some(id => stage[id] !== 1 && stage[id] !== -2)) return + if(selectedAlgoIds.some(id => !isSettled(stage[id]))) return } const rowState = rowMatchStateRef.current[rowIndex] if(!rowState) return @@ -4241,46 +4490,93 @@ const MapProject = () => { setIsLoadingInDecisionView(true) const payload = getPayloadForMatching([__row], repo) let __offset = offset || 0 + const failBridgeRow = message => { + clearRefreshRowStageSnapshot(__row.__index) + markAlgo(__row.__index, bridgeAlgoId, -2) + log({action: 'algo_failed', extras: {...getAlgoLogExtras(bridgeAlgo), error: message}}, __row.__index) + setAlert({message, severity: 'error'}) + setIsLoadingInDecisionView(false) + } return new Promise(resolve => { - bridgeRef.current?.fetchBridgeCandidates( - payload, - __offset, - isBoolean(_retired) ? _retired : retired, - (candidates) => { - if(callback) { - const newCandidates = candidates?.map(candidate => ({ - ...candidate, - results: candidate?.results?.map(result => ({ - ...result, - search_meta: { - ...result?.search_meta, - algorithm: bridgeAlgoId - } - })) - })); - callback(newCandidates, payload) - } - resolve() - }, - (response, errorMsg) => { - if(handlePreviewLimitError(response, __row.__index, bridgeAlgoId)) { + // The component calls its service's post() synchronously from here; the + // context tells that post() the row, and whether a Stop ends its waits + // (ocl_issues#2849). + const context = {rowIndex: __row.__index, isBulk, isCancelled: isBulk ? getRunStopCheck() : () => false, posted: false} + bridgeCallContextRef.current = context + try { + bridgeRef.current?.fetchBridgeCandidates( + payload, + __offset, + isBoolean(_retired) ? _retired : retired, + (candidates) => { + if(callback) { + const newCandidates = candidates?.map(candidate => ({ + ...candidate, + results: candidate?.results?.map(result => ({ + ...result, + search_meta: { + ...result?.search_meta, + algorithm: bridgeAlgoId + } + })) + })); + callback(newCandidates, payload) + } resolve() - return - } - clearRefreshRowStageSnapshot(__row.__index) - refreshMapperQuotaCache() - markAlgo(__row.__index, bridgeAlgoId, -2) - log({action: 'algo_failed', extras: getAlgoLogExtras(bridgeAlgo)}, __row.__index) - setAlert({message: response?.detail || errorMsg, severity: 'error'}) - setIsLoadingInDecisionView(false) - resolve() - }, - // ocl_online#105 Phase 5: attribution headers for the bridge $match, - // applied by the premium component (react-bridge-match). Per-row here - // (single-row payload) → scalar row_index. isBulk distinguishes a run - // call from a manual per-row bridge fetch. - attrHeaders({rowIndex: __row.__index, algorithmId: bridgeAlgoId, isRunTraffic: isBulk}) - ) + }, + (response, errorMsg) => { + if(response?.cancelled) { + markAlgo(__row.__index, bridgeAlgoId, -1) + setIsLoadingInDecisionView(false) + resolve() + return + } + // The server stayed too busy: not run, not failed. + if(response?.throttled) { + clearRefreshRowStageSnapshot(__row.__index) + markAlgo(__row.__index, bridgeAlgoId, -4) + log({action: 'algo_throttled', description: t('map_project.row_throttled'), extras: getAlgoLogExtras(bridgeAlgo)}, __row.__index) + if(isBulk) + noteThrottledRunRow(bridgeAlgoId, __row.__index) + setAlert(prev => (prev?.severity === 'error' ? prev : {message: response.detail, severity: 'warning'})) + setIsLoadingInDecisionView(false) + resolve() + return + } + if(handlePreviewLimitError(response, __row.__index, bridgeAlgoId)) { + resolve() + return + } + clearRefreshRowStageSnapshot(__row.__index) + refreshMapperQuotaCache() + markAlgo(__row.__index, bridgeAlgoId, -2) + log({action: 'algo_failed', extras: getAlgoLogExtras(bridgeAlgo)}, __row.__index) + setAlert({message: response?.detail || errorMsg, severity: 'error'}) + setIsLoadingInDecisionView(false) + resolve() + }, + // ocl_online#105 Phase 5: attribution headers for the bridge $match, + // applied by the premium component (react-bridge-match). Per-row here + // (single-row payload) → scalar row_index. isBulk distinguishes a run + // call from a manual per-row bridge fetch. + attrHeaders({rowIndex: __row.__index, algorithmId: bridgeAlgoId, isRunTraffic: isBulk}) + ) + } catch (err) { + failBridgeRow(err?.message || t('unknown_error')) + resolve() + return + } finally { + bridgeCallContextRef.current = null + } + // The component sent nothing (bridging isn't available to this user or + // project), so no callback will come: mark it n/a and settle, rather + // than leave the row "Running" and its rerank waiting for good. + if(!context.posted) { + markAlgo(__row.__index, bridgeAlgoId, -3) + log({action: 'algo_skipped', extras: {...getAlgoLogExtras(bridgeAlgo), reason: 'bridge_unavailable'}}, __row.__index) + setIsLoadingInDecisionView(false) + resolve() + } }) } @@ -4426,8 +4722,15 @@ const MapProject = () => { } else { service = service.overrideURL(oclUrl) } - const response = await service.request('GET', undefined, authToken, {query: {includeMappings: true, mappingBrief: true, mapTypes: 'SAME-AS,SAME AS,SAME_AS', verbose: true}, headers: attrHeaders({rowIndex: logRowIndex, isRunTraffic})}) - const data = response?.data + // A busy server's 429 is waited out, and a gateway error retried + // (ocl_issues#2849); a 404 is retried below. + const result = await requestWithCapacityRetry( + () => service.request('GET', undefined, authToken, {query: {includeMappings: true, mappingBrief: true, mapTypes: 'SAME-AS,SAME AS,SAME_AS', verbose: true}, headers: attrHeaders({rowIndex: logRowIndex, isRunTraffic}), handlesThrottle: true}), + {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, onCapacity: noteCapacity} + ) + if(!result.ok) + throw result.error || new Error(result.reason) + const data = result.response?.data if(data?.id) return data urlFetchCacheRef.current.delete(oclUrl) return null @@ -4481,15 +4784,23 @@ const MapProject = () => { const body = toResolve.map(({reference}) => reference.version ? {url: reference.url, version: reference.version} : {url: reference.url}) - // service.request rejects on any error, a 429 included, so the catch - // below marks these lookups failed and settles them. APIService.post - // answers a 429 with a promise that never settles, which left every - // wait on these lookups hanging (ocl_online#283). - resolvePromise = APIService.new() - .overrideURL('/$resolveReference/') - .request('POST', body, currentToken, { + // A busy server's 429 is waited out (ocl_issues#2849); any other error, + // or a server still busy at the cap, lands in the catch below, which + // marks these lookups failed and settles them. APIService.post answers + // a 429 with a promise that never settles, which left every wait on + // these lookups hanging (ocl_online#283). + resolvePromise = requestWithCapacityRetry( + () => APIService.new().overrideURL('/$resolveReference/').request('POST', body, currentToken, { headers: attrHeaders({rowIndex: logRowIndex, isRunTraffic}), query: resolveNamespace ? {namespace: resolveNamespace} : {}, + handlesThrottle: true, + }), + {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, onCapacity: noteCapacity} + ) + .then(result => { + if(!result.ok) + throw result.error || new Error(result.reason) + return result.response }) .then(async response => { const items = Array.isArray(response?.data) ? response.data : [] @@ -4574,7 +4885,7 @@ const MapProject = () => { has_filters: !isEmpty(appliedFilters || appliedFacets[rowIndex]) }) setIsLoadingInDecisionView(true) - getLookupService().get(lookupConfig?.token, null, { + const query = { includeSearchMeta: true, includeMappings: true, mappingBrief: true, @@ -4585,8 +4896,24 @@ const MapProject = () => { page: page || 1, includeRetired: includeRetired === undefined ? retired : includeRetired, ...getFacetQueryParam(appliedFilters || appliedFacets[rowIndex]), - }).then(response => { - let items = response.data + } + // service.get never settled on a 429 and resolved an error as its body, + // which then threw here. A 429 is waited out; an error says so + // (ocl_issues#2849). + requestWithCapacityRetry( + () => getLookupService().request('GET', undefined, lookupConfig?.token, {query, handlesThrottle: true}), + {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, onCapacity: noteCapacity, ...trackCapacityWait([])} + ).then(result => { + if(!result.ok) { + setIsLoadingInDecisionView(false) + if(result.reason === 'throttled') + setAlert({message: t('map_project.search_throttled'), severity: 'warning'}) + else + setAlert({message: t('map_project.search_failed', {error: result.error?.response?.data?.detail || result.error?.message || t('unknown_error')}), severity: 'error'}) + return + } + const response = result.response + let items = response.data || [] setSearchedConcepts({...searchedConcepts, [row.__index]: items}) setSearchResponse(response) setIsLoadingInDecisionView(false) @@ -4632,7 +4959,15 @@ const MapProject = () => { }) } - const request = getLookupService().get(lookupConfig?.token, null, query).then(response => { + // A failed facets call isn't cached, so the next look tries again + // (ocl_issues#2849). + const request = requestWithCapacityRetry( + () => getLookupService().request('GET', undefined, lookupConfig?.token, {query, handlesThrottle: true}), + {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, onCapacity: noteCapacity, ...trackCapacityWait([])} + ).then(result => { + if(!result.ok) + return undefined + const response = result.response const fields = response?.data?.facets?.fields || {} setFetchedFacets(prev => ({...prev, [requestKey]: fields})) if(latestFacetRequestRef.current[targetRowIndex] === requestKey) @@ -5373,7 +5708,7 @@ const MapProject = () => { const mappedSrcs = bridgeMappedSources[bridgeUrl] || [] return Boolean(repoVersion?.url) && Boolean(bridgeUrl) && mappedSrcs.length > 0 && { {abortRef.current ? t('map_project.stopping_gracefully') : t('map_project.stop_processing')} } + { + // A request is waiting out a busy server: the run is + // slower, not stuck (ocl_issues#2849). + capacityWaitRows && + } + color='warning' + variant='outlined' + size='small' + label={t('map_project.waiting_for_capacity')} + sx={{margin: '5px'}} + /> + } } { @@ -5899,6 +6247,7 @@ const MapProject = () => { candidatesScore={candidatesScore} rowIndex={rowIndex} rowStage={rowStageRef.current[rowIndex]} + capacityWait={Boolean(capacityWaitRows?.[rowIndex])} rowState={rowMatchStateRef.current[rowIndex]} conceptCache={conceptCache} targetCanonical={buildProjectContext()?.target_repo?.canonical_url} diff --git a/src/components/map-projects/__tests__/autoMatchRows.test.js b/src/components/map-projects/__tests__/autoMatchRows.test.js index 32b1d787..64e06945 100644 --- a/src/components/map-projects/__tests__/autoMatchRows.test.js +++ b/src/components/map-projects/__tests__/autoMatchRows.test.js @@ -3,8 +3,9 @@ import assert from 'node:assert/strict' import { getPreviewEligibleRowIndexes, getRowsToProcess, spendsMatchQuota, getRowCapByMatchOperations, shouldStopAIStep, - getAIRequestIdempotencyKey, getCandidatePoolFingerprint, hasCurrentAnalysis, getScispacyRowResults, getPendingRowLookups, waitForLookups, RERANK_LOOKUP_WAIT_MS, AI_LOOKUP_WAIT_MS + getAIRequestIdempotencyKey, getCandidatePoolFingerprint, hasCurrentAnalysis, getScispacyRowResults, getPendingRowLookups, waitForLookups, RERANK_LOOKUP_WAIT_MS, AI_LOOKUP_WAIT_MS, isScispacyWarmingUp, RERANK_MAX_IN_FLIGHT } from '../autoMatchRows.js' +import { createLimiter } from '../../../services/capacity.js' const rows = [ { __index: 0, label: 'zero' }, @@ -289,3 +290,43 @@ test('RERANK_LOOKUP_WAIT_MS: a hang guard, longer than the AI step\'s wait', () assert.ok(RERANK_LOOKUP_WAIT_MS >= 30000) assert.ok(RERANK_LOOKUP_WAIT_MS > AI_LOOKUP_WAIT_MS) }) + +// ocl_issues#2849: ScispaCy now goes through service.request, which rejects on +// a non-2xx. Its warm-up answers must still read as "wait", not as a failure. +test('isScispacyWarmingUp: a 503 warming_up body means the service is starting', () => { + const err = {response: {status: 503, data: {status: 'warming_up', message: 'Starting'}}} + assert.equal(isScispacyWarmingUp(err), true) +}) + +test('isScispacyWarmingUp: a 502 while the host boots counts only once warm-up was seen', () => { + const err = {response: {status: 502, data: {error: 'Bad gateway'}}} + assert.equal(isScispacyWarmingUp(err), false) + assert.equal(isScispacyWarmingUp(err, true), true) +}) + +test('isScispacyWarmingUp: other errors are not warm-up', () => { + assert.equal(isScispacyWarmingUp({response: {status: 500, data: {detail: 'boom'}}}, true), false) + assert.equal(isScispacyWarmingUp({response: {status: 429, data: {}}}, true), false) + assert.equal(isScispacyWarmingUp(new Error('Network Error'), true), false) + assert.equal(isScispacyWarmingUp(undefined), false) +}) + +// ocl_issues#2849: every $rerank goes through one limiter per tab. A finished +// 10-row batch used to fire 10 at once, on top of the rerank sweep's 2. +test('RERANK_MAX_IN_FLIGHT: a batch\'s reranks and the sweep\'s together keep at most 2 in flight', async () => { + assert.equal(RERANK_MAX_IN_FLIGHT, 2) + const limiter = createLimiter(RERANK_MAX_IN_FLIGHT) + let inFlight = 0 + let maxInFlight = 0 + const rerank = async () => { + const release = await limiter.acquire() + inFlight += 1 + maxInFlight = Math.max(maxInFlight, inFlight) + await new Promise(resolve => setTimeout(resolve, 2)) + inFlight -= 1 + release() + } + // ten per-row reranks as one batch finishes, and the sweep's two + await Promise.all([...Array(10).keys(), 'sweep-a', 'sweep-b'].map(rerank)) + assert.equal(maxInFlight, 2) +}) diff --git a/src/components/map-projects/__tests__/autosave.test.js b/src/components/map-projects/__tests__/autosave.test.js index 78b94773..01ba79f0 100644 --- a/src/components/map-projects/__tests__/autosave.test.js +++ b/src/components/map-projects/__tests__/autosave.test.js @@ -13,7 +13,7 @@ import test from 'node:test' import assert from 'node:assert/strict' -import { createAutosaveScheduler, saveOnLeave, trackSave, whenSaved, hasSaveInFlight, AUTOSAVE_DELAY_MS } from '../autosave.js' +import { createAutosaveScheduler, createLatestSender, saveOnLeave, trackSave, whenSaved, hasSaveInFlight, AUTOSAVE_DELAY_MS } from '../autosave.js' // Virtual clock standing in for setTimeout/clearTimeout. Timers scheduled // while the clock is being advanced land in a later window, as real timers @@ -304,3 +304,63 @@ test('trackSave: an older save settling does not untrack a newer one for the sam await nextTick() assert.equal(hasSaveInFlight(), false) }) + +// ── createLatestSender (ocl_issues#2849) ─────────────────────────────────── +// The logs POST sends the whole log each time. With retries, an older POST +// that waited out a 429 could land after a newer one and overwrite it; so one +// POST at a time, and triggers during it coalesce into one more, which reads +// the log as it is then. + +const deferred = () => { + let resolve + const promise = new Promise(r => { resolve = r }) + return {promise, resolve} +} + +test('createLatestSender: one trigger sends once', async () => { + let sends = 0 + const sender = createLatestSender(async () => { sends += 1 }) + await sender.trigger() + assert.equal(sends, 1) +}) + +test('createLatestSender: triggers while a send is in flight coalesce into one more send', async () => { + const gates = [deferred(), deferred()] + let sends = 0 + const sender = createLatestSender(() => gates[sends++].promise) + const first = sender.trigger() + sender.trigger() + sender.trigger() + const last = sender.trigger() + assert.equal(sends, 1) + gates[0].resolve() + await new Promise(resolve => setTimeout(resolve, 0)) + assert.equal(sends, 2) + gates[1].resolve() + // every trigger's promise settles once the log is sent as it stood last + await Promise.all([first, last]) + assert.equal(sends, 2) +}) + +test('createLatestSender: a failed send doesn\'t stop the pending one, and nothing rejects', async () => { + let sends = 0 + const sender = createLatestSender(async () => { + sends += 1 + if(sends === 1) { + await new Promise(resolve => setTimeout(resolve, 5)) + throw new Error('boom') + } + }) + const first = sender.trigger() + const second = sender.trigger() + await Promise.all([first, second]) + assert.equal(sends, 2) +}) + +test('createLatestSender: after it settles, the next trigger sends again', async () => { + let sends = 0 + const sender = createLatestSender(async () => { sends += 1 }) + await sender.trigger() + await sender.trigger() + assert.equal(sends, 2) +}) diff --git a/src/components/map-projects/__tests__/matchBatch.test.js b/src/components/map-projects/__tests__/matchBatch.test.js index 505d3dd9..7692c0f9 100644 --- a/src/components/map-projects/__tests__/matchBatch.test.js +++ b/src/components/map-projects/__tests__/matchBatch.test.js @@ -12,7 +12,8 @@ import test from 'node:test' import assert from 'node:assert/strict' -import { formatRowNumbers, isRetryableMatchError, runMatchBatch, requestSingleMatch } from '../matchBatch.js' +import { formatRowNumbers, isRetryableMatchError, runMatchBatch, requestSingleMatch, runWithConcurrency } from '../matchBatch.js' +import { createCapacityGate } from '../../../services/capacity.js' import { createLatestRequestGate } from '../aiVisibility.js' const NO_WAIT = {baseDelayMs: 0} @@ -286,3 +287,211 @@ test('requestSingleMatch: a preview-limit 403 passes its body through unchanged assert.equal(result.previewLimit, true) assert.deepEqual(result.errorBody, limit) }) + +// ── Waiting out a busy server (OpenConceptLab/ocl_issues#2849) ──────────────── +// A 429 used to fail the batch at once. Now it waits for Retry-After (plus +// jitter) and resumes; a batch still throttled after the long cap is marked +// throttled (-4, "not run, retry"), never failed. + +const throttled = () => httpError(429, {}) +const withRetryAfter = (err, seconds) => { err.response.headers = {'retry-after': String(seconds)}; return err } +const virtualClock = () => { + const clock = {t: 0} + clock.now = () => clock.t + clock.sleep = async ms => { clock.t += ms } + return clock +} + +test('runMatchBatch: a 429 waits for Retry-After and the batch then succeeds; its rows are never marked failed', async () => { + const rec = recorder() + const clock = virtualClock() + const waits = [] + let calls = 0 + const send = async () => { + calls += 1 + if(calls === 1) throw withRetryAfter(throttled(), 12) + return ok([{row: {__index: 0}, results: []}, {row: {__index: 1}, results: []}]) + } + + const data = await runMatchBatch({ + rowIndexes: [0, 1], send, ...rec, + onWait: info => waits.push(info.reason), onWaitEnd: () => waits.push('end'), + retryOptions: {now: clock.now, sleep: clock.sleep, random: () => 0}, + }) + + assert.equal(data.length, 2) + assert.equal(calls, 2) + assert.equal(clock.t, 12000) + assert.deepEqual(rec.stages, {0: [0, 1], 1: [0, 1]}) + assert.deepEqual(rec.failed, []) + assert.deepEqual(rec.finished, [0, 1]) + assert.deepEqual(waits, ['throttled', 'end']) +}) + +test('runMatchBatch: a batch still throttled after the cap is marked throttled (-4), not failed', async () => { + const rec = recorder() + const clock = virtualClock() + const throttledRows = [] + + const data = await runMatchBatch({ + rowIndexes: [5, 6], send: async () => { throw withRetryAfter(throttled(), 60) }, ...rec, + onRowThrottled: (index, info) => throttledRows.push([index, info.attempts > 1]), + retryOptions: {now: clock.now, sleep: clock.sleep, random: () => 0, maxWaitMs: 5 * 60 * 1000}, + }) + + assert.deepEqual(data, []) + assert.deepEqual(rec.stages, {5: [0, -4], 6: [0, -4]}) + assert.deepEqual(rec.failed, []) + assert.deepEqual(rec.finished, []) + assert.deepEqual(throttledRows, [[5, true], [6, true]]) +}) + +test('runMatchBatch: Stop during a 429 wait puts the rows back to not run and sends nothing more', async () => { + const rec = recorder() + let calls = 0 + let stopped = false + const send = async () => { + calls += 1 + setTimeout(() => { stopped = true }, 20) + throw withRetryAfter(throttled(), 600) + } + const started = Date.now() + + const data = await runMatchBatch({rowIndexes: [0, 1], send, ...rec, shouldStop: () => stopped, retryOptions: {pollMs: 5}}) + + assert.deepEqual(data, []) + assert.equal(calls, 1) + assert.ok(Date.now() - started < 1000) + assert.deepEqual(rec.stages, {0: [0, -1], 1: [0, -1]}) + assert.deepEqual(rec.failed, []) +}) + +test('runMatchBatch: a 503 is still retried, then fails the rows with its status', async () => { + const rec = recorder() + let calls = 0 + await runMatchBatch({rowIndexes: [0], send: async () => { calls += 1; throw httpError(503) }, ...rec, retryOptions: NO_WAIT}) + assert.equal(calls, 3) + assert.deepEqual(rec.stages, {0: [0, -2]}) + assert.equal(rec.failed[0].status, 503) +}) + +test('runMatchBatch: a 429 on one batch holds back another batch sharing the gate', async () => { + const clock = virtualClock() + const gate = createCapacityGate({now: clock.now}) + const sentAt = {a: [], b: []} + const sendA = async () => { + sentAt.a.push(clock.t) + if(sentAt.a.length === 1) throw withRetryAfter(throttled(), 30) + return ok([]) + } + const sendB = async () => { sentAt.b.push(clock.t); return ok([]) } + const common = {retryOptions: {gate, now: clock.now, sleep: clock.sleep, random: () => 0}} + + await runMatchBatch({rowIndexes: [0], send: sendA, ...recorder(), ...common}) + gate.pause(30000) + await runMatchBatch({rowIndexes: [1], send: sendB, ...recorder(), ...common}) + + assert.deepEqual(sentAt.a, [0, 30000]) + assert.deepEqual(sentAt.b, [60000]) +}) + +test('requestSingleMatch: a 429 waits and then returns the response', async () => { + const clock = virtualClock() + let calls = 0 + const result = await requestSingleMatch(async () => { + calls += 1 + if(calls === 1) throw withRetryAfter(throttled(), 5) + return ok([{ row: { __index: 3 }, results: [] }]) + }, { retryOptions: {now: clock.now, sleep: clock.sleep, random: () => 0} }) + assert.equal(result.ok, true) + assert.equal(calls, 2) + assert.equal(clock.t, 5000) +}) + +test('requestSingleMatch: still throttled after its cap, it reports throttled, not a failure', async () => { + const clock = virtualClock() + const result = await requestSingleMatch(async () => { throw withRetryAfter(throttled(), 60) }, { retryOptions: {now: clock.now, sleep: clock.sleep, random: () => 0} }) + assert.equal(result.ok, false) + assert.equal(result.throttled, true) + assert.equal(result.previewLimit, false) + assert.equal(result.errorBody.status, 429) + // a person is waiting on the row panel: 5 minutes, not a run's 30 + assert.ok(clock.t <= 5 * 60 * 1000 && clock.t >= 4 * 60 * 1000, `waited ${clock.t}`) +}) + +// ── the bulk scheduler ───────────────────────────────────────────────────────── + +test('runWithConcurrency: at most `concurrency` in flight, every item run once', async () => { + let inFlight = 0 + let maxInFlight = 0 + const ran = [] + const done = await runWithConcurrency([1, 2, 3, 4, 5, 6, 7], { + concurrency: 3, + run: async item => { + inFlight += 1 + maxInFlight = Math.max(maxInFlight, inFlight) + await new Promise(resolve => setTimeout(resolve, 3)) + ran.push(item) + inFlight -= 1 + }, + }) + assert.equal(done, true) + assert.equal(maxInFlight, 3) + assert.deepEqual(ran.sort(), [1, 2, 3, 4, 5, 6, 7]) +}) + +test('runWithConcurrency: sends no new batch while the gate is paused, then resumes', async () => { + const gate = createCapacityGate() + const started = Date.now() + const dispatchedAt = {} + const done = await runWithConcurrency([1, 2, 3, 4], { + concurrency: 2, + gate, + pollMs: 5, + run: async item => { + dispatchedAt[item] = Date.now() - started + // the first batch is throttled: the server asks everyone to wait 80 ms + if(item === 1) + gate.pause(80) + await new Promise(resolve => setTimeout(resolve, 5)) + }, + }) + assert.equal(done, true) + assert.ok(dispatchedAt[1] < 40) + for(const item of [2, 3, 4]) + assert.ok(dispatchedAt[item] >= 75, `batch ${item} went out at ${dispatchedAt[item]} ms, during the pause`) +}) + +test('runWithConcurrency: Stop during a pause returns at once, and sends nothing more', async () => { + const gate = createCapacityGate() + let stopped = false + const ran = [] + const started = Date.now() + setTimeout(() => { stopped = true }, 20) + const done = await runWithConcurrency([1, 2, 3], { + concurrency: 1, + gate, + pollMs: 5, + shouldAbort: () => stopped, + run: async item => { ran.push(item); gate.pause(60000) }, + }) + assert.equal(done, false) + assert.deepEqual(ran, [1]) + assert.ok(Date.now() - started < 1000) +}) + +test('runWithConcurrency: shouldSkipRest drops the queue but lets the batches in flight finish', async () => { + let skip = false + const finished = [] + const done = await runWithConcurrency([1, 2, 3, 4, 5], { + concurrency: 2, + shouldSkipRest: () => skip, + run: async item => { + if(item === 2) skip = true + await new Promise(resolve => setTimeout(resolve, 3)) + finished.push(item) + }, + }) + assert.equal(done, true) + assert.deepEqual(finished.sort(), [1, 2]) +}) diff --git a/src/components/map-projects/__tests__/rowProgress.test.js b/src/components/map-projects/__tests__/rowProgress.test.js new file mode 100644 index 00000000..48bb7cfc --- /dev/null +++ b/src/components/map-projects/__tests__/rowProgress.test.js @@ -0,0 +1,59 @@ +/** + * The row panel's progress chip (Candidates.jsx). + * + * ocl_issues#2849: a row whose request is waiting out a busy server says so, + * and a row the server stayed too busy for says "retry", never "failed". + * + * Run with: npm test + */ + +import test from 'node:test' +import assert from 'node:assert/strict' + +import { getRowProgressLabel } from '../rowProgress.js' + +const ALGOS = [{id: 'ocl-semantic'}, {id: 'ocl-bridge'}] +const t = key => key + +test('getRowProgressLabel: nothing for a row with no stages yet', () => { + assert.deepEqual(getRowProgressLabel(undefined, ALGOS, {t}), {label: false}) +}) + +test('getRowProgressLabel: not started, running, waiting and partial, as before', () => { + assert.deepEqual(getRowProgressLabel({'ocl-semantic': -1, 'ocl-bridge': -1}, ALGOS, {t}), {label: 'Not started', status: 'idle'}) + assert.deepEqual(getRowProgressLabel({'ocl-semantic': 0, 'ocl-bridge': -1}, ALGOS, {t}), {label: 'Running: ocl-semantic...', status: 'running'}) + assert.deepEqual(getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': -1}, ALGOS, {t}), {label: 'Waiting: ocl-bridge', status: 'waiting'}) + assert.deepEqual(getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': -2}, ALGOS, {t}), {label: 'Partially completed', status: 'partial'}) +}) + +test('getRowProgressLabel: all done gives no label', () => { + assert.equal(getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': 1}, ALGOS, {t}).label, undefined) +}) + +test('getRowProgressLabel: a row waiting for capacity says so, over "Running"', () => { + assert.deepEqual( + getRowProgressLabel({'ocl-semantic': 0, 'ocl-bridge': -1}, ALGOS, {t, capacityWait: true}), + {label: 'map_project.waiting_for_capacity', status: 'capacity_wait'}, + ) +}) + +test('getRowProgressLabel: a row waiting for capacity says so even once its algorithms are done (its rerank is waiting)', () => { + assert.equal(getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': 1}, ALGOS, {t, capacityWait: true}).status, 'capacity_wait') +}) + +test('getRowProgressLabel: a throttled row asks for a retry, not "failed" or "partial"', () => { + assert.deepEqual( + getRowProgressLabel({'ocl-semantic': -4, 'ocl-bridge': 1}, ALGOS, {t}), + {label: 'map_project.row_throttled', status: 'throttled'}, + ) +}) + +test('getRowProgressLabel: while other algorithms still run, the row shows them, not throttled', () => { + assert.equal(getRowProgressLabel({'ocl-semantic': -4, 'ocl-bridge': 0}, ALGOS, {t}).status, 'running') + assert.equal(getRowProgressLabel({'ocl-semantic': -4, 'ocl-bridge': -1}, ALGOS, {t}).status, 'waiting') +}) + +test('getRowProgressLabel: an algorithm that can\'t run for this user (-3, n/a) counts as done', () => { + assert.equal(getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': -3}, ALGOS, {t}).label, undefined) + assert.equal(getRowProgressLabel({'ocl-semantic': -2, 'ocl-bridge': -3}, ALGOS, {t}).status, 'partial') +}) diff --git a/src/components/map-projects/autoMatchRows.js b/src/components/map-projects/autoMatchRows.js index 22df09c6..77ab3514 100644 --- a/src/components/map-projects/autoMatchRows.js +++ b/src/components/map-projects/autoMatchRows.js @@ -122,6 +122,16 @@ export const hasCurrentAnalysis = (analyses, fingerprint) => { // the row's index, not its position in the run (ocl_online#258). export const getScispacyRowResults = (responseData, rowIndex) => responseData?.[rowIndex] || [] +// Whether a failed ScispaCy call means the service is still starting: it +// answers 503 {status: 'warming_up'} while its host boots, then a 502 until +// the app is up. The row waits for it rather than failing. +export const isScispacyWarmingUp = (err, seenWarmingUp = false) => { + const data = err?.response?.data + if(data?.status === 'warming_up') + return true + return Boolean(seenWarmingUp && err?.response?.status === 502) +} + // The in-flight $lookups for a row's concepts. Rerank waits for them, and so // does a bulk run's AI step, which can reach a row whose rerank was skipped // (after a rerank quota stop), so the AI sees the looked-up concepts. @@ -133,6 +143,11 @@ export const getPendingRowLookups = (rowState, inFlightLookups) => // How long a bulk run's AI step waits for a row's lookups before going ahead. export const AI_LOOKUP_WAIT_MS = 15000 +// $rerank calls in flight at once, per tab, whatever fired them: the rerank +// sweep, or the per-row reranks as batches finish (ocl_issues#2849). Each ties +// up an API worker for seconds, and a finished 10-row batch used to fire ten. +export const RERANK_MAX_IN_FLIGHT = 2 + // How long rerank waits for a row's lookups before going ahead without them // (ocl_online#283). A guard against a lookup that never settles, longer than // the AI step's wait: a rerank that goes ahead early drops the concepts still diff --git a/src/components/map-projects/autosave.js b/src/components/map-projects/autosave.js index 30cac9a5..067c57b8 100644 --- a/src/components/map-projects/autosave.js +++ b/src/components/map-projects/autosave.js @@ -141,3 +141,35 @@ export const installUnloadGuard = () => { event.returnValue = '' }) } + +/** + * One send at a time; triggers that arrive during it coalesce into one more + * send, which reads its payload then (ocl_issues#2849). For the project logs + * POST, which sends the whole log: once it waits out a 429, an older POST + * could otherwise land after a newer one and overwrite it. trigger() resolves, + * never rejects, once nothing is left to send. + */ +export const createLatestSender = send => { + let chain = null + let pending = false + const loop = async () => { + do { + pending = false + try { + await send() + } catch (_) { + // Best effort: the next change sends the whole log again. + } + } while(pending) + chain = null + } + return { + trigger: () => { + if(chain) + pending = true + else + chain = loop() + return chain + }, + } +} diff --git a/src/components/map-projects/constants.jsx b/src/components/map-projects/constants.jsx index fd8a7113..a24146fc 100644 --- a/src/components/map-projects/constants.jsx +++ b/src/components/map-projects/constants.jsx @@ -66,6 +66,8 @@ export const ES_BATCH_SIZE = 50 export const CANDIDATES_LIMIT = 30 export const ROW_STAGES = { + // the server stayed too busy: not run, retry (ocl_issues#2849) + '-4': 'throttled', '-3': 'na', '-2': 'failed', '-1': 'not_started', diff --git a/src/components/map-projects/matchBatch.js b/src/components/map-projects/matchBatch.js index dd257322..8d143439 100644 --- a/src/components/map-projects/matchBatch.js +++ b/src/components/map-projects/matchBatch.js @@ -6,18 +6,22 @@ * candidates. The caller's `send` must reject instead (service.request does), * so a failure is caught here, retried when it's transient, and reported per * row. + * + * A 429 waits for the server's Retry-After and resumes (ocl_issues#2849): a + * busy server makes a run slower, not broken. */ import { isPreviewLimitError } from './previewLimits.js' -import { isTransientNetworkError, retryWithBackoff } from '../../services/retry.js' +import { isRetryableError, requestWithCapacityRetry } from '../../services/capacity.js' -// The gateway answers these while the API rolls out or a task restarts. A 500 -// comes from the API itself, and retrying it only adds load. -const RETRYABLE_STATUSES = new Set([502, 503, 504]) +// How long the row panel waits out a 429 before showing the row as throttled. +// A person is watching it, so less than a run's 30 minutes. +export const INTERACTIVE_WAIT_CAP_MS = 5 * 60 * 1000 export const MATCH_RETRY_OPTIONS = {maxRetries: 2, baseDelayMs: 3000, backoffFactor: 4, jitterFactor: 0.25} -export const isRetryableMatchError = err => - isTransientNetworkError(err) || RETRYABLE_STATUSES.has(err?.response?.status) +// Network errors and 502/503/504. A 429 isn't an error to retry: it's waited +// out, however many times it comes. +export const isRetryableMatchError = isRetryableError const getFailure = (err, attempts) => ({ error: err?.response?.data?.detail || err?.message || 'unknown', @@ -27,13 +31,19 @@ const getFailure = (err, attempts) => ({ }) /** - * Mark the batch's rows running, send it (retrying transient failures), then - * mark each row done (1) or failed (-2). Never rejects. + * Mark the batch's rows running, send it, then mark each row done (1), failed + * (-2) or throttled (-4). Never rejects. + * + * A network error or a 502/503/504 is retried a bounded number of times, then + * fails the rows. A 429 waits for Retry-After, however often it comes, until + * the long cap; a batch still refused then is throttled (-4): "not run, retry", + * never failed. Waits on a gate shared with the run's other requests hold them + * all back together. * - * shouldStop is checked before every retry, including after the backoff - * sleep. A batch stopped there (the run was cancelled, or another batch hit - * the match quota) sends nothing more and puts its rows back to not run (-1), - * like the rows the run never reached. + * shouldStop is checked before every send and during every wait. A batch + * stopped there (the run was cancelled, or another batch hit the match quota) + * sends nothing more and puts its rows back to not run (-1), like the rows the + * run never reached. * * @param {object} opts * @param {number[]} opts.rowIndexes the batch's row __index values @@ -41,57 +51,91 @@ const getFailure = (err, attempts) => ({ * @param {function} opts.setStage (rowIndex, stage) => void * @param {function} opts.onRowFinished rowIndex => void * @param {function} opts.onRowFailed (rowIndex, {error, status, attempts, previewLimit}) => void + * @param {function} [opts.onRowThrottled] (rowIndex, {attempts, waitedMs}) => void * @param {function} [opts.onPreviewLimit] err => void, for a preview-limit 403 (never retried) - * @param {function} [opts.shouldStop] () => boolean; true skips any further retry - * @param {object} [opts.retryOptions] overrides MATCH_RETRY_OPTIONS - * @returns {Promise} the batch's $match results, or [] when it failed or stopped + * @param {function} [opts.onWait] info => void, as each wait starts (see requestWithCapacityRetry) + * @param {function} [opts.onWaitEnd] () => void, as each wait ends + * @param {function} [opts.shouldStop] () => boolean; true skips any further send + * @param {object} [opts.retryOptions] overrides MATCH_RETRY_OPTIONS; also takes gate, onCapacity and maxWaitMs + * @returns {Promise} the batch's $match results, or [] when it failed, was throttled or stopped */ export const runMatchBatch = async ({ - rowIndexes, send, setStage, onRowFinished, onRowFailed, onPreviewLimit, shouldStop = () => false, retryOptions = {}, + rowIndexes, send, setStage, onRowFinished, onRowFailed, onRowThrottled, onPreviewLimit, onWait, onWaitEnd, + shouldStop = () => false, retryOptions = {}, }) => { rowIndexes.forEach(index => setStage(index, 0)) - let attempts = 0 - let lastError - let stopped = false - // retryWithBackoff asks this only when it would otherwise retry. - const stopBeforeRetry = () => { - stopped = shouldStop() - return stopped - } - let response - try { - response = await retryWithBackoff(attempt => { - // The run may have stopped while this batch slept in its backoff. - if(attempt > 0 && stopBeforeRetry()) - throw lastError - attempts = attempt + 1 - return send(attempt) - }, { - ...MATCH_RETRY_OPTIONS, - ...retryOptions, - isCancelled: stopBeforeRetry, - isRetryable: isRetryableMatchError, - onAttemptFailed: err => { lastError = err }, + const result = await requestWithCapacityRetry(send, { + ...MATCH_RETRY_OPTIONS, + ...retryOptions, + isCancelled: shouldStop, + onWait, + onWaitEnd, + }) + if(result.ok) { + rowIndexes.forEach(index => { + setStage(index, 1) + onRowFinished(index) }) - } catch (err) { - if(stopped) { - rowIndexes.forEach(index => setStage(index, -1)) - return [] - } - const failure = getFailure(err, attempts) + return result.response?.data || [] + } + if(result.reason === 'cancelled') { + rowIndexes.forEach(index => setStage(index, -1)) + return [] + } + if(result.reason === 'throttled') { rowIndexes.forEach(index => { - setStage(index, -2) - onRowFailed(index, failure) + setStage(index, -4) + onRowThrottled?.(index, {attempts: result.attempts, waitedMs: result.waitedMs}) }) - if(failure.previewLimit) - onPreviewLimit?.(err) return [] } + const failure = getFailure(result.error, result.attempts) rowIndexes.forEach(index => { - setStage(index, 1) - onRowFinished(index) + setStage(index, -2) + onRowFailed(index, failure) }) - return response?.data || [] + if(failure.previewLimit) + onPreviewLimit?.(result.error) + return [] +} + +/** + * The bulk Auto Match scheduler: runs each item through run(item) with at most + * concurrency in flight. While the gate is paused (a request got a 429) it + * sends nothing new, and waits for the gate to reopen or a request in flight + * to settle, so a throttled run doesn't drain its queue into more refusals. + * + * shouldAbort stops it at once (the requests in flight are left to finish on + * their own) and returns false. shouldSkipRest drops the rest of the queue but + * waits for the requests in flight. Otherwise returns true once every item ran. + */ +export const runWithConcurrency = async (items, { + concurrency, run, shouldAbort = () => false, shouldSkipRest = () => false, gate = null, pollMs, +}) => { + const queue = items.slice() + const active = new Set() + while(queue.length || active.size) { + while(queue.length && active.size < concurrency) { + if(shouldAbort()) + return false + if(shouldSkipRest()) { + queue.length = 0 + break + } + if(gate?.isPaused()) + break + const item = queue.shift() + const promise = run(item).finally(() => active.delete(promise)) + active.add(promise) + } + if(queue.length && active.size < concurrency && gate?.isPaused()) { + await Promise.race([gate.wait(shouldAbort, {pollMs}), ...active]) + continue + } + if(active.size) + await Promise.race(active) + } + return true } /** @@ -116,27 +160,29 @@ export const formatRowNumbers = (rowIndexes, {maxRanges = 10} = {}) => { /** * Send one single-row $match (the row panel's match and "load more"), - * retrying transient failures like a bulk batch (ocl_online#283). Never - * rejects. After the retries it gives an errorBody for the row's existing - * failure path: the server's JSON when it carries a detail or an error_code - * (so a preview limit still opens its dialog), else {detail, status}. + * retrying transient failures like a bulk batch (ocl_online#283) and waiting + * out 429s for up to INTERACTIVE_WAIT_CAP_MS (ocl_issues#2849). Never rejects. + * After the retries it gives an errorBody for the row's existing failure + * path: the server's JSON when it carries a detail or an error_code (so a + * preview limit still opens its dialog), else {detail, status}. throttled is + * true when the server was still busy at the cap. * * @param {function} send attempt => Promise; must reject on failure - * @param {object} [opts.retryOptions] overrides MATCH_RETRY_OPTIONS - * @returns {Promise<{ok: true, response: object}|{ok: false, errorBody: object, previewLimit: boolean}>} + * @param {object} [opts.retryOptions] overrides MATCH_RETRY_OPTIONS; also takes gate, onWait, onWaitEnd, onCapacity + * @returns {Promise<{ok: true, response: object}|{ok: false, errorBody: object, previewLimit: boolean, throttled: boolean}>} */ export const requestSingleMatch = async (send, { retryOptions = {} } = {}) => { - try { - const response = await retryWithBackoff(send, {...MATCH_RETRY_OPTIONS, ...retryOptions, isRetryable: isRetryableMatchError}) - return { ok: true, response } - } catch (err) { - const data = err?.response?.data - const hasServerBody = data && typeof data === 'object' && !Array.isArray(data) && (data.detail || data.error_code) - const { error, status } = getFailure(err, 0) - return { - ok: false, - errorBody: hasServerBody ? data : { detail: error, status }, - previewLimit: isPreviewLimitError(err), - } + const result = await requestWithCapacityRetry(send, {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, ...MATCH_RETRY_OPTIONS, ...retryOptions}) + if(result.ok) + return { ok: true, response: result.response } + const err = result.error + const data = err?.response?.data + const hasServerBody = data && typeof data === 'object' && !Array.isArray(data) && (data.detail || data.error_code) + const { error, status } = getFailure(err, 0) + return { + ok: false, + errorBody: hasServerBody ? data : { detail: error, status }, + previewLimit: isPreviewLimitError(err), + throttled: result.reason === 'throttled', } } diff --git a/src/components/map-projects/rowProgress.js b/src/components/map-projects/rowProgress.js new file mode 100644 index 00000000..924b6c09 --- /dev/null +++ b/src/components/map-projects/rowProgress.js @@ -0,0 +1,47 @@ +/** + * The row panel's progress chip: which of the row's algorithms is running or + * still to run. {label: false} before the row has stages; no label once every + * algorithm is done. + * + * capacityWait: a request for this row (an algorithm or its rerank) is waiting + * out a busy server, which the chip says instead (ocl_issues#2849). A row the + * server stayed too busy for (-4) asks for a retry: it wasn't run, it didn't + * fail. + */ +export const getRowProgressLabel = (stageMap, algos, { t, capacityWait = false } = {}) => { + if(stageMap === undefined) + return {label: false} + if(capacityWait) + return {label: t('map_project.waiting_for_capacity'), status: 'capacity_wait'} + if(!stageMap) + return {label: 'Preparing...', status: 'partial'} + + const stages = algos.map(k => stageMap[k.id]); + + if (!stages.length || stages.every(v => v === -1)) { + return { label: 'Not started', status: 'idle' }; + } + + const runningIndex = stages.findIndex(v => v === 0); + if (runningIndex !== -1) { + return { + label: `Running: ${algos[runningIndex].id}...`, + status: 'running', + }; + } + + // -3: the algorithm isn't available to this user, so there's nothing to wait for. + if (stages.every(v => v === 1 || v === -3)) { + return true + } + + const waitingIndex = stages.findIndex(v => v === -1) + if(waitingIndex !== -1) { + return {label: `Waiting: ${algos[waitingIndex].id}`, status: 'waiting'} // AutoMatch Bulk + } + + if(stages.some(v => v === -4)) + return {label: t('map_project.row_throttled'), status: 'throttled'} + + return { label: 'Partially completed', status: 'partial' }; +} diff --git a/src/i18n/locales/en/translations.json b/src/i18n/locales/en/translations.json index 88556175..388a011b 100644 --- a/src/i18n/locales/en/translations.json +++ b/src/i18n/locales/en/translations.json @@ -589,6 +589,13 @@ "auto_match_rows_failed": "Matching failed for {{count}} input row(s), so they're missing candidates: {{details}}. Rows count from the first data row of your file. Run Auto Match on those rows again to retry them.", "auto_match_rows_failed_algorithm": "{{algorithm}} on row(s) {{rows}}", "match_request_failed": "Couldn't get candidates ({{error}}). Try again.", + "waiting_for_capacity": "Waiting for capacity: results may take longer due to demand", + "row_throttled": "Server busy: not matched yet. Retry later", + "algorithm_throttled": "The server is busy, so this row wasn't matched this time. Nothing failed: run it again later.", + "auto_match_rows_throttled": "The server was busy, so {{count}} input row(s) weren't matched: {{details}}. Nothing failed. Run Auto Match on those rows again later to finish them.", + "rerank_step": "Reranking", + "search_throttled": "The server is busy, so the search didn't run. Try again in a few minutes.", + "search_failed": "Couldn't search ({{error}}). Try again.", "run_ai_analysis": "Run AI Analysis", "run_ai_analysis_note": "Note: Enabling this feature will run AI Analysis on results of each row. This has direct cost implications.", "decision": "Decision", diff --git a/src/i18n/locales/es/translations.json b/src/i18n/locales/es/translations.json index 4c74e7c2..9e549a5f 100644 --- a/src/i18n/locales/es/translations.json +++ b/src/i18n/locales/es/translations.json @@ -553,6 +553,13 @@ "auto_match_rows_failed": "La coincidencia falló en {{count}} fila(s) de entrada, así que les faltan candidatos: {{details}}. Las filas se cuentan desde la primera fila de datos de tu archivo. Vuelve a ejecutar Auto Match en esas filas para reintentarlo.", "auto_match_rows_failed_algorithm": "{{algorithm}} en la(s) fila(s) {{rows}}", "match_request_failed": "No se pudieron obtener candidatos ({{error}}). Inténtalo de nuevo.", + "waiting_for_capacity": "Esperando capacidad: los resultados pueden tardar más debido a la demanda", + "row_throttled": "Servidor ocupado: aún sin procesar. Reintenta más tarde", + "algorithm_throttled": "El servidor está ocupado, así que esta fila no se procesó esta vez. Nada falló: vuelve a ejecutarla más tarde.", + "auto_match_rows_throttled": "El servidor estaba ocupado, así que {{count}} fila(s) de entrada no se procesaron: {{details}}. Nada falló. Vuelve a ejecutar Auto Match en esas filas más tarde para terminarlas.", + "rerank_step": "Reordenamiento", + "search_throttled": "El servidor está ocupado, así que la búsqueda no se ejecutó. Inténtalo de nuevo en unos minutos.", + "search_failed": "No se pudo buscar ({{error}}). Inténtalo de nuevo.", "run_ai_analysis": "Ejecutar Análisis de IA", "run_ai_analysis_note": "Nota: Habilitar esta función ejecutará Análisis de IA en los resultados de cada fila. Esto tiene implicaciones de costo directas.", "decision": "Decisión", diff --git a/src/i18n/locales/zh/translations.json b/src/i18n/locales/zh/translations.json index 8552a31a..3eb220ce 100644 --- a/src/i18n/locales/zh/translations.json +++ b/src/i18n/locales/zh/translations.json @@ -578,6 +578,13 @@ "auto_match_rows_failed": "{{count}} 个输入行匹配失败,缺少候选项:{{details}}。行号从文件的第一个数据行开始计数。请对这些行重新运行自动匹配以重试。", "auto_match_rows_failed_algorithm": "{{algorithm}}:第 {{rows}} 行", "match_request_failed": "无法获取候选项({{error}})。请重试。", + "waiting_for_capacity": "正在等待容量:由于需求量大,结果可能需要更长时间", + "row_throttled": "服务器繁忙:尚未匹配。请稍后重试", + "algorithm_throttled": "服务器繁忙,本次未能匹配此行。没有出错:请稍后重新运行。", + "auto_match_rows_throttled": "服务器繁忙,{{count}} 个输入行未能匹配:{{details}}。没有出错。请稍后对这些行重新运行自动匹配以完成匹配。", + "rerank_step": "重新排序", + "search_throttled": "服务器繁忙,搜索未能运行。请几分钟后重试。", + "search_failed": "无法搜索({{error}})。请重试。", "run_ai_analysis": "运行 AI 分析", "run_ai_analysis_note": "注意:启用此功能将对每行的结果运行 AI 分析。这会产生直接的成本影响。", "decision": "决策", diff --git a/src/services/APIService.js b/src/services/APIService.js index c945b5f9..e7a8e89a 100644 --- a/src/services/APIService.js +++ b/src/services/APIService.js @@ -156,6 +156,9 @@ class APIService { }; } + // config.handlesThrottle: the caller waits out a 429 itself + // (services/capacity.js), so the full-screen "Too many requests" countdown, + // which would also cover the Stop button, stays closed (ocl_issues#2849). request(method, data, token, config) { let headers = {}; let query = {}; @@ -164,9 +167,10 @@ class APIService { query = get(config, 'query', {}); } let request = this.getRequest(method, data, token, headers, query); - request = {...request, ...omit(config, ['headers', 'query'])}; + request = {...request, ...omit(config, ['headers', 'query', 'handlesThrottle'])}; return axios(request).catch(error => { - handleAPIError(error); + if(!(config?.handlesThrottle && error?.response?.status === 429)) + handleAPIError(error); return Promise.reject(error); }); }; diff --git a/src/services/__tests__/attribution.test.js b/src/services/__tests__/attribution.test.js index 79e12b25..a9362e80 100644 --- a/src/services/__tests__/attribution.test.js +++ b/src/services/__tests__/attribution.test.js @@ -212,3 +212,22 @@ test('user cancel wins over a quota stop', () => { const r = summarizeRunCompletion({ rowStages: {0: {'ocl-search': 1}}, rowIndices: [0], algoIds: ['ocl-search'], aborted: true, stoppedForQuota: true }) assert.equal(r.completion_status, 'cancelled') }) + +// ocl_issues#2849: a row the server was too busy to match (-4) wasn't run, it +// didn't fail. It counts as neither, and a run with such rows is partial. +test('throttled rows (-4) count as neither completed nor failed', () => { + const rowStages = { 0: {'ocl-search': 1}, 1: {'ocl-search': -4}, 2: {'ocl-search': -2} } + const r = summarizeRunCompletion({ rowStages, rowIndices: [0, 1, 2], algoIds: ['ocl-search'] }) + assert.deepEqual(r, { completed_rows: 1, failed_rows: 1, completion_status: 'partial' }) +}) + +test('a run whose rows were all throttled is partial, not failed', () => { + const rowStages = { 0: {'ocl-search': -4}, 1: {'ocl-search': -4, 'ocl-semantic': -2} } + const r = summarizeRunCompletion({ rowStages, rowIndices: [0, 1], algoIds: ALGOS }) + assert.deepEqual(r, { completed_rows: 0, failed_rows: 0, completion_status: 'partial' }) +}) + +test('a user cancel wins over throttled rows', () => { + const r = summarizeRunCompletion({ rowStages: {0: {'ocl-search': -4}}, rowIndices: [0], algoIds: ['ocl-search'], aborted: true }) + assert.equal(r.completion_status, 'cancelled') +}) diff --git a/src/services/__tests__/capacity.test.js b/src/services/__tests__/capacity.test.js new file mode 100644 index 00000000..c3df31a2 --- /dev/null +++ b/src/services/__tests__/capacity.test.js @@ -0,0 +1,485 @@ +/** + * Waiting out a busy server (OpenConceptLab/ocl_issues#2849). + * + * A 429 means "slow down", not "failed": the call waits for its Retry-After + * (plus 0-50% jitter) and tries again, until a long cap. A 502/503/504 or a + * network error is retried a bounded number of times and then reported as an + * error. Stop ends any wait. Requests to one server share a gate, so a 429 on + * one pauses the others. + * + * The tests drive a virtual clock: sleep advances it instead of waiting. + * + * Run with: npm test + */ + +import test from 'node:test' +import assert from 'node:assert/strict' + +import { + CAPACITY_WAIT_CAP_MS, + createCapacityGate, + createLimiter, + getCapacityHeaders, + getRetryAfterMs, + requestWithCapacityRetry, + sleepUnlessCancelled, +} from '../capacity.js' + +const networkError = () => Object.assign(new Error('Network Error'), {isAxiosError: true, code: 'ERR_NETWORK'}) +const httpError = (status, {data = {}, headers = {}} = {}) => Object.assign(new Error(`Request failed with status code ${status}`), { + isAxiosError: true, code: 'ERR_BAD_RESPONSE', response: {status, data, headers}, +}) +const throttled = (retryAfter, extra = {}) => httpError(429, {headers: retryAfter === undefined ? {} : {'retry-after': String(retryAfter)}, ...extra}) +const ok = (data = [], headers = {}) => ({status: 200, data, headers}) + +const virtualClock = () => { + const clock = {t: 0, sleeps: []} + clock.now = () => clock.t + clock.sleep = async ms => { + clock.sleeps.push(ms) + clock.t += ms + } + return clock +} + +// Sends answer from a script, one entry per attempt: an error to throw, or a +// response to return. +const scripted = (...answers) => { + const sent = [] + const send = async attempt => { + sent.push(attempt) + const answer = answers[Math.min(sent.length - 1, answers.length - 1)] + if(answer instanceof Error) throw answer + return answer + } + return {send, sent} +} + +// ── getRetryAfterMs ────────────────────────────────────────────────────────── + +test('getRetryAfterMs: seconds from the Retry-After header, whatever its case', () => { + assert.equal(getRetryAfterMs({headers: {'retry-after': '12'}}), 12000) + assert.equal(getRetryAfterMs({headers: {'Retry-After': '3'}}), 3000) + // axios' AxiosHeaders has a get() + assert.equal(getRetryAfterMs({headers: {get: key => (key.toLowerCase() === 'retry-after' ? '7' : undefined)}}), 7000) +}) + +test('getRetryAfterMs: an HTTP date counts from now', () => { + const now = Date.parse('2026-09-30T12:00:00Z') + assert.equal(getRetryAfterMs({headers: {'retry-after': 'Wed, 30 Sep 2026 12:00:20 GMT'}}, {now: () => now}), 20000) + // a date already past means "now" + assert.equal(getRetryAfterMs({headers: {'retry-after': 'Wed, 30 Sep 2026 11:00:00 GMT'}}, {now: () => now}), 0) +}) + +test('getRetryAfterMs: the body\'s retry_after when the header is missing', () => { + assert.equal(getRetryAfterMs({headers: {}, data: {retry_after: 15}}), 15000) +}) + +test('getRetryAfterMs: null when the server gave no delay, or garbage', () => { + assert.equal(getRetryAfterMs({headers: {}}), null) + assert.equal(getRetryAfterMs({headers: {'retry-after': 'soon'}}), null) + assert.equal(getRetryAfterMs({headers: {'retry-after': '-4'}}), null) + assert.equal(getRetryAfterMs(undefined), null) +}) + +// ── getCapacityHeaders ────────────────────────────────────────────────────── + +test('getCapacityHeaders: reads the X-OCL-Capacity-* headers', () => { + const headers = { + 'x-ocl-capacity-decision': 'shadow-refused', + 'x-ocl-capacity-limit': '4', + 'x-ocl-capacity-in-flight': '5', + 'x-ocl-capacity-tier': 'preview', + 'x-ocl-capacity-tier-limit': '2', + 'x-ocl-capacity-tier-in-flight': '3', + 'x-ocl-capacity-suggested-concurrency': '1', + 'x-ocl-capacity-suggested-spacing-ms': '500', + } + assert.deepEqual(getCapacityHeaders({headers}), { + decision: 'shadow-refused', limit: 4, inFlight: 5, tier: 'preview', tierLimit: 2, tierInFlight: 3, + suggestedConcurrency: 1, suggestedSpacingMs: 500, + }) +}) + +test('getCapacityHeaders: null when the response carries none', () => { + assert.equal(getCapacityHeaders({headers: {'content-type': 'application/json'}}), null) + assert.equal(getCapacityHeaders(undefined), null) +}) + +test('getCapacityHeaders: only the headers present', () => { + assert.deepEqual(getCapacityHeaders({headers: {'X-OCL-Capacity-Decision': 'admitted'}}), {decision: 'admitted'}) +}) + +// ── requestWithCapacityRetry ──────────────────────────────────────────────── + +test('requestWithCapacityRetry: a 429 waits its Retry-After plus 0-50% jitter, then succeeds', async () => { + const clock = virtualClock() + const {send, sent} = scripted(throttled(10), ok(['row'])) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, random: () => 0.5}) + + assert.equal(result.ok, true) + assert.deepEqual(result.response.data, ['row']) + assert.deepEqual(sent, [0, 1]) + // 10 s + 25% jitter (random 0.5 × 50%) + assert.equal(clock.t, 12500) + assert.equal(result.waitedMs, 12500) +}) + +test('requestWithCapacityRetry: the jitter stays within 0-50% of Retry-After', async () => { + for(const random of [0, 0.999]) { + const clock = virtualClock() + const {send} = scripted(throttled(8), ok()) + await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, random: () => random}) + assert.ok(clock.t >= 8000 && clock.t < 12000, `waited ${clock.t}`) + } +}) + +test('requestWithCapacityRetry: 429s keep waiting past the error-retry budget, and aren\'t failures', async () => { + const clock = virtualClock() + const {send, sent} = scripted(throttled(5), throttled(5), throttled(5), throttled(5), throttled(5), ok()) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, random: () => 0, maxRetries: 2}) + + assert.equal(result.ok, true) + assert.equal(sent.length, 6) + assert.equal(clock.t, 25000) +}) + +test('requestWithCapacityRetry: after the long cap a throttled call ends "throttled", not "error"', async () => { + const clock = virtualClock() + const {send, sent} = scripted(throttled(60)) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, random: () => 0}) + + assert.equal(result.ok, false) + assert.equal(result.reason, 'throttled') + assert.equal(result.error.response.status, 429) + // it waited up to the cap, never past it + assert.ok(clock.t <= CAPACITY_WAIT_CAP_MS, `waited ${clock.t}`) + assert.ok(clock.t >= CAPACITY_WAIT_CAP_MS - 60000, `waited ${clock.t}`) + assert.equal(sent.length, clock.t / 60000 + 1) +}) + +test('requestWithCapacityRetry: the cap is 30 minutes', () => { + assert.equal(CAPACITY_WAIT_CAP_MS, 30 * 60 * 1000) +}) + +test('requestWithCapacityRetry: a Retry-After past the cap ends "throttled" at once, without waiting', async () => { + const clock = virtualClock() + // a daily limit: come back tomorrow + const {send, sent} = scripted(throttled(86400)) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep}) + + assert.equal(result.reason, 'throttled') + assert.equal(sent.length, 1) + assert.equal(clock.t, 0) +}) + +test('requestWithCapacityRetry: a 429 without Retry-After backs off from 5 s, doubling to at most 60 s', async () => { + const clock = virtualClock() + const {send} = scripted(throttled(), throttled(), throttled(), throttled(), throttled(), throttled(), ok()) + + await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, random: () => 0}) + + assert.equal(clock.t, 5000 + 10000 + 20000 + 40000 + 60000 + 60000) +}) + +test('requestWithCapacityRetry: a 503 is retried twice with backoff, then reported as an error', async () => { + const clock = virtualClock() + const {send, sent} = scripted(httpError(503)) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, random: () => 0.5}) + + assert.equal(result.ok, false) + assert.equal(result.reason, 'error') + assert.equal(result.error.response.status, 503) + assert.equal(result.attempts, 3) + assert.equal(sent.length, 3) + // 3 s, then 12 s (factor 4), no jitter at random 0.5 + assert.equal(clock.t, 15000) +}) + +test('requestWithCapacityRetry: a 503 with Retry-After waits that long before its retry', async () => { + const clock = virtualClock() + const {send} = scripted(httpError(503, {headers: {'retry-after': '20'}}), ok()) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, random: () => 0}) + + assert.equal(result.ok, true) + assert.equal(clock.t, 20000) +}) + +test('requestWithCapacityRetry: a network failure is retried twice, then reported as an error', async () => { + const clock = virtualClock() + const {send, sent} = scripted(networkError()) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep}) + + assert.equal(result.reason, 'error') + assert.equal(result.error.message, 'Network Error') + assert.equal(sent.length, 3) +}) + +test('requestWithCapacityRetry: a transient failure that recovers keeps the response', async () => { + const clock = virtualClock() + const {send} = scripted(networkError(), httpError(502), ok(['x'])) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep}) + + assert.equal(result.ok, true) + assert.equal(result.attempts, 3) +}) + +test('requestWithCapacityRetry: a 500 or a 4xx is an error at once', async () => { + for(const status of [500, 400, 403, 404]) { + const clock = virtualClock() + const {send, sent} = scripted(httpError(status)) + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep}) + assert.equal(result.reason, 'error', `status ${status}`) + assert.equal(sent.length, 1, `status ${status}`) + assert.equal(clock.t, 0, `status ${status}`) + } +}) + +test('requestWithCapacityRetry: a caller can widen what counts as retryable', async () => { + const clock = virtualClock() + const {send, sent} = scripted(httpError(404), ok()) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, isRetryable: err => err?.response?.status === 404}) + + assert.equal(result.ok, true) + assert.equal(sent.length, 2) +}) + +test('requestWithCapacityRetry: Stop during a 429 wait ends it at once, "cancelled", with nothing more sent', async () => { + let stopped = false + const clock = virtualClock() + const sleep = async ms => { + await clock.sleep(ms) + if(clock.t >= 3000) stopped = true + } + const {send, sent} = scripted(throttled(600), ok()) + + const result = await requestWithCapacityRetry(send, {now: clock.now, sleep, isCancelled: () => stopped}) + + assert.equal(result.ok, false) + assert.equal(result.reason, 'cancelled') + assert.equal(sent.length, 1) + // polled every 250 ms: stopped within one poll of the Stop, not after 600 s + assert.ok(clock.t <= 3250, `waited ${clock.t}`) +}) + +test('requestWithCapacityRetry: Stop during an error backoff ends it too', async () => { + let stopped = false + const {send, sent} = scripted(networkError(), ok()) + + const result = await requestWithCapacityRetry(send, { + isCancelled: () => stopped, + sleep: async () => { stopped = true }, + }) + + assert.equal(result.reason, 'cancelled') + assert.equal(sent.length, 1) +}) + +test('requestWithCapacityRetry: a call cancelled before it starts sends nothing', async () => { + const {send, sent} = scripted(ok()) + const result = await requestWithCapacityRetry(send, {isCancelled: () => true}) + assert.equal(result.reason, 'cancelled') + assert.equal(sent.length, 0) +}) + +test('requestWithCapacityRetry: with real timers, Stop breaks a long wait within a poll', async () => { + let stopped = false + setTimeout(() => { stopped = true }, 30) + const started = Date.now() + + const result = await requestWithCapacityRetry(scripted(throttled(600)).send, {isCancelled: () => stopped, pollMs: 10}) + + assert.equal(result.reason, 'cancelled') + assert.ok(Date.now() - started < 1000) +}) + +test('requestWithCapacityRetry: onWait and onWaitEnd bracket each wait, with the reason and the delay', async () => { + const clock = virtualClock() + const events = [] + const {send} = scripted(throttled(10), httpError(503), ok()) + + await requestWithCapacityRetry(send, { + now: clock.now, sleep: clock.sleep, random: () => 0, + onWait: info => events.push(['wait', info.reason, info.delayMs, info.status]), + onWaitEnd: () => events.push(['end']), + }) + + assert.deepEqual(events, [['wait', 'throttled', 10000, 429], ['end'], ['wait', 'error', 2250, 503], ['end']]) +}) + +test('requestWithCapacityRetry: onCapacity gets the capacity headers of a success and of a 429', async () => { + const clock = virtualClock() + const seen = [] + const {send} = scripted( + throttled(1, {headers: {'retry-after': '1', 'x-ocl-capacity-decision': 'refused'}}), + ok([], {'x-ocl-capacity-decision': 'admitted', 'x-ocl-capacity-suggested-concurrency': '2'}), + ) + + await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, onCapacity: headers => seen.push(headers)}) + + assert.deepEqual(seen, [{decision: 'refused'}, {decision: 'admitted', suggestedConcurrency: 2}]) +}) + +test('requestWithCapacityRetry: a send that throws a non-HTTP error is an error, not retried', async () => { + const {send, sent} = scripted(new TypeError('x is undefined')) + const result = await requestWithCapacityRetry(send, {}) + assert.equal(result.reason, 'error') + assert.equal(sent.length, 1) +}) + +// ── the gate: one 429 pauses every request to that server ──────────────────── + +test('createCapacityGate: pause extends, never shortens, the pause', () => { + const clock = virtualClock() + const gate = createCapacityGate({now: clock.now}) + assert.equal(gate.isPaused(), false) + gate.pause(10000) + gate.pause(4000) + assert.equal(gate.pausedForMs(), 10000) + clock.t = 6000 + gate.pause(8000) + assert.equal(gate.pausedForMs(), 8000) + clock.t = 14000 + assert.equal(gate.isPaused(), false) +}) + +test('createCapacityGate: wait sleeps until the gate reopens, and Stop ends it', async () => { + const clock = virtualClock() + const gate = createCapacityGate({now: clock.now}) + gate.pause(5000) + assert.equal(await gate.wait(() => false, {sleep: clock.sleep}), true) + assert.equal(clock.t, 5000) + + gate.pause(60000) + let stopped = false + const sleep = async ms => { await clock.sleep(ms); stopped = clock.t >= 6000 } + assert.equal(await gate.wait(() => stopped, {sleep}), false) + assert.ok(clock.t < 7000) +}) + +test('requestWithCapacityRetry: a 429 on one request holds back another that shares the gate', async () => { + const clock = virtualClock() + const gate = createCapacityGate({now: clock.now}) + const first = scripted(throttled(20), ok()) + const second = scripted(ok()) + + await requestWithCapacityRetry(first.send, {gate, now: clock.now, sleep: clock.sleep, random: () => 0}) + // the first call slept through the pause; start a second while it's still paused + gate.pause(20000) + const startedAt = clock.t + const events = [] + const result = await requestWithCapacityRetry(second.send, { + gate, now: clock.now, sleep: clock.sleep, random: () => 0, + onWait: info => events.push(info.reason), + }) + + assert.equal(result.ok, true) + assert.equal(clock.t - startedAt, 20000) + assert.deepEqual(events, ['paused']) +}) + +test('requestWithCapacityRetry: a 429 pauses the shared gate for its Retry-After', async () => { + const clock = virtualClock() + const gate = createCapacityGate({now: clock.now}) + let pausedDuringWait = null + const sleep = async ms => { + if(pausedDuringWait === null) pausedDuringWait = gate.pausedForMs() + await clock.sleep(ms) + } + + await requestWithCapacityRetry(scripted(throttled(30), ok()).send, {gate, now: clock.now, sleep, random: () => 0}) + + assert.equal(pausedDuringWait, 30000) +}) + +test('requestWithCapacityRetry: time waiting on the gate counts toward the cap', async () => { + const clock = virtualClock() + const gate = createCapacityGate({now: clock.now}) + gate.pause(CAPACITY_WAIT_CAP_MS + 60000) + const {send, sent} = scripted(ok()) + + const result = await requestWithCapacityRetry(send, {gate, now: clock.now, sleep: clock.sleep}) + + assert.equal(result.reason, 'throttled') + assert.equal(sent.length, 0) +}) + +// ── sleepUnlessCancelled ──────────────────────────────────────────────────── + +test('sleepUnlessCancelled: true after the full sleep, false as soon as it is cancelled', async () => { + const clock = virtualClock() + assert.equal(await sleepUnlessCancelled(1000, () => false, {sleep: clock.sleep}), true) + assert.equal(clock.t, 1000) + assert.equal(await sleepUnlessCancelled(1000, () => clock.t >= 1500, {sleep: clock.sleep}), false) + assert.equal(clock.t, 1500) +}) + +// ── createLimiter: at most N in flight ────────────────────────────────────── + +test('createLimiter: at most 2 in flight; the rest queue and run in order', async () => { + const limiter = createLimiter(2) + let inFlight = 0 + let maxInFlight = 0 + const order = [] + const task = async id => { + const release = await limiter.acquire() + inFlight += 1 + maxInFlight = Math.max(maxInFlight, inFlight) + order.push(id) + await new Promise(resolve => setTimeout(resolve, 5)) + inFlight -= 1 + release() + } + + await Promise.all([1, 2, 3, 4, 5, 6, 7].map(task)) + + assert.equal(maxInFlight, 2) + assert.deepEqual(order, [1, 2, 3, 4, 5, 6, 7]) + assert.equal(limiter.active(), 0) +}) + +test('createLimiter: a queued caller cancelled before its turn gets null and takes no slot', async () => { + const limiter = createLimiter(1) + const release = await limiter.acquire() + let stopped = false + const queued = limiter.acquire({isCancelled: () => stopped, pollMs: 5}) + stopped = true + assert.equal(await queued, null) + release() + assert.equal(limiter.active(), 0) + const next = await limiter.acquire() + assert.equal(limiter.active(), 1) + next() +}) + +test('createLimiter: releasing twice frees one slot only', async () => { + const limiter = createLimiter(2) + const a = await limiter.acquire() + await limiter.acquire() + a() + a() + assert.equal(limiter.active(), 1) +}) + +test('requestWithCapacityRetry: a throttled wait carries the 429\'s capacity headers, for the row log', async () => { + const clock = virtualClock() + const waits = [] + const {send} = scripted( + throttled(2, {headers: {'retry-after': '2', 'x-ocl-capacity-decision': 'refused', 'x-ocl-capacity-tier': 'preview'}}), + throttled(2), + ok(), + ) + + await requestWithCapacityRetry(send, {now: clock.now, sleep: clock.sleep, onWait: info => waits.push(info.capacity)}) + + assert.deepEqual(waits, [{decision: 'refused', tier: 'preview'}, null]) +}) diff --git a/src/services/attribution.js b/src/services/attribution.js index 75766b46..8502cd39 100644 --- a/src/services/attribution.js +++ b/src/services/attribution.js @@ -125,11 +125,12 @@ export const buildConfigSnapshot = ({ * Derive AutomatchRun completion counts + status from the per-row algo stages * captured during a run (oclapi2 PATCH /auto-match-runs//). * - * Stage vocabulary (oclmap rowStageRef): -2 failed, -1 not run yet, 0 running, - * 1 done. A row is **completed** when ANY attempted algo finished (1) and - * **failed** when EVERY attempted algo failed (-2). Rows that never ran - * (-1/undefined — e.g. a run cancelled before reaching them) or were still in - * flight (0) count as NEITHER, so a cancelled run never over-reports + * Stage vocabulary (oclmap rowStageRef): -4 throttled, -2 failed, -1 not run + * yet, 0 running, 1 done. A row is **completed** when ANY attempted algo + * finished (1) and **failed** when EVERY attempted algo failed (-2). Rows that + * never ran (-1/undefined — e.g. a run cancelled before reaching them), were + * still in flight (0) or were throttled (-4: the server stayed too busy, so + * the row wasn't run) count as NEITHER, so a cancelled run never over-reports * completed_rows (ocl_online#115 review). * * @param {object} [opts] @@ -148,10 +149,12 @@ export const summarizeRunCompletion = ({ } = {}) => { let completed = 0 let failed = 0 + let throttled = 0 rowIndices.forEach(idx => { const stage = rowStages[idx] || {} const attempted = algoIds.map(id => stage[id]).filter(s => s !== undefined && s !== -1) if(!attempted.length) return + if(attempted.some(s => s === -4)) throttled += 1 if(attempted.some(s => s === 1)) completed += 1 else if(attempted.every(s => s === -2)) failed += 1 }) @@ -161,6 +164,9 @@ export const summarizeRunCompletion = ({ // Running out of match quota isn't a user cancel: the rows that finished are // kept. completion_status has no quota value yet, so it's recorded as partial. else if(stoppedForQuota) completion_status = 'partial' + // Rows the server was too busy to match can be run again: not a failure + // (ocl_issues#2849). + else if(throttled) completion_status = 'partial' else if(completed === 0) completion_status = 'failed' else if(completed < total) completion_status = 'partial' return { completed_rows: completed, failed_rows: failed, completion_status } diff --git a/src/services/capacity.js b/src/services/capacity.js new file mode 100644 index 00000000..387c3bee --- /dev/null +++ b/src/services/capacity.js @@ -0,0 +1,314 @@ +// Waiting out a busy server (OpenConceptLab/ocl_issues#2849). Dependency-free +// so node's test runner can load it. +// +// A 429 means "slow down", not "failed": the request waits for the server's +// Retry-After, plus 0-50% at random so held-back requests don't all resend at +// once, and tries again until it has waited CAPACITY_WAIT_CAP_MS in all. Then +// it ends "throttled", which callers show as "not run, retry", never as a +// failure. A 502/503/504 or a network error is retried a bounded number of +// times, then ends "error". Stop ends any wait. + +import { isTransientNetworkError } from './retry.js' + +export const CAPACITY_WAIT_CAP_MS = 30 * 60 * 1000 + +const THROTTLE_JITTER = 0.5 +// A 429 with Retry-After: 0 must not turn into a tight resend loop. +const MIN_THROTTLE_WAIT_MS = 1000 +// A 429 without Retry-After waits 5 s, doubling up to a minute. +const DEFAULT_THROTTLE_WAIT_MS = 5000 +const MAX_DEFAULT_THROTTLE_WAIT_MS = 60000 +// A 503's Retry-After is honoured, but it's an error retry: keep it short. +const MAX_ERROR_RETRY_AFTER_MS = 60000 +// How often a wait checks for Stop. Stop sets a flag (abortRef), with no event +// to listen to. +const POLL_MS = 250 + +// The gateway answers these while the API rolls out or a task restarts. A 500 +// comes from the API itself, and retrying it only adds load. +const RETRYABLE_STATUSES = new Set([502, 503, 504]) + +export const isRetryableError = err => + isTransientNetworkError(err) || RETRYABLE_STATUSES.has(err?.response?.status) + +const defaultSleep = ms => new Promise(resolve => setTimeout(resolve, ms)) + +const getHeader = (headers, name) => { + if(!headers) + return undefined + if(typeof headers.get === 'function') { + const value = headers.get(name) + if(value !== undefined && value !== null) + return value + } + const key = Object.keys(headers).find(k => k.toLowerCase() === name) + return key === undefined ? undefined : headers[key] +} + +/** + * How long a response asks the client to wait, in ms: the Retry-After header + * (seconds, or an HTTP date), else the body's retry_after (seconds). null when + * it gives none. + */ +export const getRetryAfterMs = (response, { now = Date.now } = {}) => { + const header = getHeader(response?.headers, 'retry-after') + if(header !== undefined && header !== null && `${header}`.trim() !== '') { + const text = `${header}`.trim() + if(/^-?\d+(\.\d+)?$/.test(text)) { + const seconds = Number(text) + return seconds >= 0 ? seconds * 1000 : null + } + if(/[a-z]/i.test(text)) { + const at = Date.parse(text) + if(Number.isFinite(at)) + return Math.max(at - now(), 0) + } + return null + } + const seconds = response?.data?.retry_after + return typeof seconds === 'number' && Number.isFinite(seconds) && seconds >= 0 ? seconds * 1000 : null +} + +const CAPACITY_HEADERS = { + decision: ['x-ocl-capacity-decision', String], + limit: ['x-ocl-capacity-limit', Number], + inFlight: ['x-ocl-capacity-in-flight', Number], + tier: ['x-ocl-capacity-tier', String], + tierLimit: ['x-ocl-capacity-tier-limit', Number], + tierInFlight: ['x-ocl-capacity-tier-in-flight', Number], + suggestedConcurrency: ['x-ocl-capacity-suggested-concurrency', Number], + suggestedSpacingMs: ['x-ocl-capacity-suggested-spacing-ms', Number], +} + +/** + * The server's X-OCL-Capacity-* headers (OpenConceptLab/ocl_online#275), or + * null when the response carries none. Read and logged only for now; adapting + * to the suggested concurrency is OpenConceptLab/ocl_online#340. + */ +export const getCapacityHeaders = response => { + const capacity = {} + Object.entries(CAPACITY_HEADERS).forEach(([field, [name, parse]]) => { + const value = getHeader(response?.headers, name) + if(value === undefined || value === null || value === '') + return + const parsed = parse(value) + if(parse === Number && !Number.isFinite(parsed)) + return + capacity[field] = parsed + }) + return Object.keys(capacity).length ? capacity : null +} + +/** + * Sleeps ms, checking isCancelled every pollMs. Resolves true after the full + * sleep, false as soon as it is cancelled. + */ +export const sleepUnlessCancelled = async (ms, isCancelled = () => false, { sleep = defaultSleep, pollMs = POLL_MS } = {}) => { + let remaining = ms + while(remaining > 0) { + if(isCancelled()) + return false + const step = Math.min(remaining, pollMs) + await sleep(step) + remaining -= step + } + return !isCancelled() +} + +/** + * One per server, shared by the requests a tab sends it. A 429 on any of them + * pauses the gate for its Retry-After, and the others (and the Auto Match + * scheduler) hold back until it reopens. + */ +export const createCapacityGate = ({ now = Date.now } = {}) => { + let resumeAt = 0 + const pausedForMs = () => Math.max(resumeAt - now(), 0) + return { + pause: ms => { resumeAt = Math.max(resumeAt, now() + ms) }, + pausedForMs, + isPaused: () => pausedForMs() > 0, + // Resolves true once the gate is open, false if cancelled first. + wait: async (isCancelled = () => false, { sleep, pollMs } = {}) => { + while(pausedForMs() > 0) { + if(!(await sleepUnlessCancelled(pausedForMs(), isCancelled, { sleep, pollMs }))) + return false + } + return true + }, + } +} + +/** + * Sends a request, waiting out 429s and retrying transient failures. Never + * rejects. + * + * @param {function} send attempt => Promise; must reject on a non-2xx (service.request does) + * @param {object} [opts] + * @param {object} [opts.gate] a createCapacityGate() shared with other requests to the same server + * @param {function} [opts.isCancelled] () => boolean; Stop. Checked before each send and during every wait + * @param {number} [opts.maxWaitMs] how long to wait in all before ending "throttled" + * @param {number} [opts.maxRetries] retries for a network error or 502/503/504 (429s don't count) + * @param {function} [opts.isRetryable] err => boolean; replaces the network-or-gateway check + * @param {function} [opts.onWait] ({reason: 'throttled'|'paused'|'error', delayMs, status, retryAfterMs, capacity}) => void + * @param {function} [opts.onWaitEnd] () => void, after each wait, however it ended + * @param {function} [opts.onCapacity] capacityHeaders => void, for each response that carries them + * @returns {Promise<{ok: true, response, attempts, waitedMs}|{ok: false, reason: 'throttled'|'cancelled'|'error', error, attempts, waitedMs}>} + */ +export const requestWithCapacityRetry = async (send, { + gate = null, + isCancelled = () => false, + maxWaitMs = CAPACITY_WAIT_CAP_MS, + maxRetries = 2, + baseDelayMs = 3000, + backoffFactor = 4, + jitterFactor = 0.25, + isRetryable = isRetryableError, + onWait, + onWaitEnd, + onCapacity, + now = Date.now, + random = Math.random, + sleep = defaultSleep, + pollMs = POLL_MS, +} = {}) => { + let attempts = 0 + let errorRetries = 0 + let throttles = 0 + let waitedMs = 0 + let lastError + const end = fields => ({ ...fields, attempts, waitedMs }) + const cancelled = () => end({ ok: false, reason: 'cancelled', error: lastError }) + const reportCapacity = response => { + const capacity = onCapacity ? getCapacityHeaders(response) : null + if(capacity) + onCapacity(capacity) + } + const waitFor = async (delayMs, info) => { + onWait?.({ ...info, delayMs }) + const startedAt = now() + const slept = await sleepUnlessCancelled(delayMs, isCancelled, { sleep, pollMs }) + waitedMs += now() - startedAt + onWaitEnd?.() + return slept + } + + for(;;) { + if(isCancelled()) + return cancelled() + + if(gate?.isPaused()) { + const pauseMs = gate.pausedForMs() + if(waitedMs + pauseMs > maxWaitMs) + return end({ ok: false, reason: 'throttled', error: lastError }) + onWait?.({ reason: 'paused', delayMs: pauseMs, status: null, retryAfterMs: pauseMs }) + const startedAt = now() + let open = await gate.wait(isCancelled, { sleep, pollMs }) + // Spread out the requests the pause held back. + if(open) + open = await sleepUnlessCancelled(pauseMs * THROTTLE_JITTER * random(), isCancelled, { sleep, pollMs }) + waitedMs += now() - startedAt + onWaitEnd?.() + if(!open) + return cancelled() + continue + } + + attempts += 1 + let response + try { + response = await send(attempts - 1) + } catch (err) { + lastError = err + reportCapacity(err?.response) + if(err?.response?.status === 429) { + const retryAfterMs = getRetryAfterMs(err.response, { now }) + const baseMs = retryAfterMs === null ? + Math.min(DEFAULT_THROTTLE_WAIT_MS * 2 ** throttles, MAX_DEFAULT_THROTTLE_WAIT_MS) : + Math.max(retryAfterMs, MIN_THROTTLE_WAIT_MS) + throttles += 1 + const delayMs = baseMs * (1 + random() * THROTTLE_JITTER) + if(waitedMs + delayMs > maxWaitMs) + return end({ ok: false, reason: 'throttled', error: err }) + gate?.pause(baseMs) + if(!(await waitFor(delayMs, { reason: 'throttled', status: 429, retryAfterMs, capacity: getCapacityHeaders(err.response) }))) + return cancelled() + continue + } + if(errorRetries < maxRetries && isRetryable(err)) { + const retryAfterMs = getRetryAfterMs(err?.response, { now }) + const delayMs = retryAfterMs === null ? + baseDelayMs * backoffFactor ** errorRetries * (1 - jitterFactor + random() * jitterFactor * 2) : + Math.min(retryAfterMs, MAX_ERROR_RETRY_AFTER_MS) * (1 + random() * THROTTLE_JITTER) + errorRetries += 1 + if(!(await waitFor(delayMs, { reason: 'error', status: err?.response?.status ?? null, retryAfterMs }))) + return cancelled() + continue + } + return end({ ok: false, reason: 'error', error: err }) + } + reportCapacity(response) + return end({ ok: true, response }) + } +} + +/** + * At most max callers hold a slot at once; the rest queue in order. acquire + * resolves a release function, or null when isCancelled turned true before + * the caller's turn. + */ +export const createLimiter = max => { + let active = 0 + const waiting = [] + const makeRelease = () => { + let released = false + return () => { + if(released) + return + released = true + active -= 1 + grantNext() + } + } + const grantNext = () => { + while(active < max && waiting.length) { + const next = waiting.shift() + if(next.isCancelled()) { + next.resolve(null) + continue + } + active += 1 + next.resolve(makeRelease()) + } + } + const acquire = ({ isCancelled, pollMs = POLL_MS } = {}) => { + const cancelledNow = isCancelled || (() => false) + if(cancelledNow()) + return Promise.resolve(null) + if(active < max && !waiting.length) { + active += 1 + return Promise.resolve(makeRelease()) + } + return new Promise(resolve => { + let timer = null + const entry = { + isCancelled: cancelledNow, + resolve: value => { + if(timer) + clearInterval(timer) + resolve(value) + }, + } + if(isCancelled) + timer = setInterval(() => { + if(!cancelledNow()) + return + const at = waiting.indexOf(entry) + if(at !== -1) + waiting.splice(at, 1) + entry.resolve(null) + }, pollMs) + waiting.push(entry) + }) + } + return { acquire, active: () => active, waiting: () => waiting.length } +} From ec304f00d4e475617d73a65db13df0f496f87855 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Tue, 29 Sep 2026 23:06:49 -0400 Subject: [PATCH 02/14] OpenConceptLab/ocl_issues#2849 | Preview accounts keep at most 2 $match batches in flight; early access stays at 5 requestLimits.js now caps by the user's highest tier instead of a single non-core cap: core users and staff keep the form's full range, early access 10 rows x 5 requests, and preview (or no tier) 10 rows x 2. Many preview users at once could fill the server's matching capacity. A preview project has at most 25 rows, 3 batches of 10, so 2 in flight costs one more batch's time per run. The form's limit and helper text follow the same tier. Kept as its own commit: the number is still being confirmed. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- .../map-projects/ConfigurationForm.jsx | 4 +- src/components/map-projects/MapProject.jsx | 13 +-- .../map-projects/MultiAlgoSelector.jsx | 15 ++-- .../__tests__/requestLimits.test.js | 80 +++++++++++++------ src/components/map-projects/requestLimits.js | 43 ++++++---- 5 files changed, 100 insertions(+), 55 deletions(-) diff --git a/src/components/map-projects/ConfigurationForm.jsx b/src/components/map-projects/ConfigurationForm.jsx index dc9782e3..bb9630d4 100644 --- a/src/components/map-projects/ConfigurationForm.jsx +++ b/src/components/map-projects/ConfigurationForm.jsx @@ -61,7 +61,7 @@ const VisuallyHiddenInput = styled('input')({ const deriveCanonicalUrl = relativeUrl => relativeUrl ? `https://ns.openconceptlab.org${relativeUrl}` : '' -const ConfigurationForm = ({ project, handleFileUpload, file, owner, setOwner, name, setName, description, setDescription, repo, onRepoChange, repoVersion, setRepoVersion, versions, mappedSources, targetSourcesFromRows, algosSelected, setAlgosSelected, sx, algos, validColumns, columns, isValidColumnValue, updateColumn, configure, setConfigure, columnVisibilityModel, setColumnVisibilityModel, onSave, isSaving, candidatesScore, onScoreChange, includeDefaultFilter, setIncludeDefaultFilter, filters, setFilters, locales, isLoadingLocales, setAIAssistantColumns, AIAssistantColumns, inAIAssistantGroup, lookupConfig, setLookupConfig, encoderModel, setEncoderModel, isCoreUser, fullRequestLimits, canSelectAIModel, canSetAIOutputLocale, canBridge, canScispacy, canUseOrgProjects=true, canUseCustomAlgorithms=true, promptTemplates, promptTemplate, onPromptTemplateChange, AIModels, AIModel, setAIModel, namespace, setNamespace, promptOutputLocale, setPromptOutputLocale, inputLocale, setInputLocale, oclLocales, useLexicalVariants, setUseLexicalVariants }) => { +const ConfigurationForm = ({ project, handleFileUpload, file, owner, setOwner, name, setName, description, setDescription, repo, onRepoChange, repoVersion, setRepoVersion, versions, mappedSources, targetSourcesFromRows, algosSelected, setAlgosSelected, sx, algos, validColumns, columns, isValidColumnValue, updateColumn, configure, setConfigure, columnVisibilityModel, setColumnVisibilityModel, onSave, isSaving, candidatesScore, onScoreChange, includeDefaultFilter, setIncludeDefaultFilter, filters, setFilters, locales, isLoadingLocales, setAIAssistantColumns, AIAssistantColumns, inAIAssistantGroup, lookupConfig, setLookupConfig, encoderModel, setEncoderModel, isCoreUser, requestLimits, canSelectAIModel, canSetAIOutputLocale, canBridge, canScispacy, canUseOrgProjects=true, canUseCustomAlgorithms=true, promptTemplates, promptTemplate, onPromptTemplateChange, AIModels, AIModel, setAIModel, namespace, setNamespace, promptOutputLocale, setPromptOutputLocale, inputLocale, setInputLocale, oclLocales, useLexicalVariants, setUseLexicalVariants }) => { const { t } = useTranslation(); const user = getCurrentUser() const isLLMAlgoNotAllowed = !repoVersion?.match_algorithms?.includes('llm') @@ -438,7 +438,7 @@ const ConfigurationForm = ({ project, handleFileUpload, file, owner, setOwner, n onChange={setAlgosSelected} repo={repoVersion} isCoreUser={isCoreUser} - fullRequestLimits={fullRequestLimits} + requestLimits={requestLimits} /> { isCoreUser && diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index 3ba62471..caf8257e 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -101,7 +101,7 @@ import Concept from './Concept' import ImportToCollection from './ImportToCollection' import ProjectLogs from './ProjectLogs'; import { useAlgos, ensureConceptIdentity } from './algorithms' -import { applyRequestSettings, getRequestSettings, hasFullRequestLimits } from './requestLimits' +import { applyRequestSettings, getRequestLimits, getRequestSettings } from './requestLimits' import { countQualityBuckets, createBestCandidateScoreCache, filterRowsByQualityBucket, getRowQualities } from './rowQuality' import AutoMatchDialog from './AutoMatchDialog' import PreviewLimitDialog from './PreviewLimitDialog' @@ -608,8 +608,9 @@ const MapProject = () => { const OCL_ONLINE_API_URL = window.OCL_ONLINE_API_URL || process.env.OCL_ONLINE_API_URL const inAIAssistantGroup = Boolean(hasCapability(user, 'users.mapper_ai_assistant') && AI_ASSISTANT_API_URL) const isCoreUser = hasAuthGroup(user, 'core_user') - // Batch size and concurrent requests per algorithm (ocl_online#274) - const fullRequestLimits = hasFullRequestLimits(user) + // Batch size and concurrent requests per algorithm, by the user's tier + // (ocl_online#274, ocl_issues#2849) + const requestLimits = getRequestLimits(user) const mapperPreview = getMapperPreview() // Only staff choose the AI model and prompt template (canSelectAIModel gates // both pickers). Core, early-access and unlimited-AI users keep the output @@ -2298,7 +2299,7 @@ const MapProject = () => { // Function to handle concurrency const processWithConcurrency = async (_repo, algo, _rows) => { // A project's saved settings run within this user's limits. - const { batchSize, concurrentRequests } = getRequestSettings(algo, fullRequestLimits) + const { batchSize, concurrentRequests } = getRequestSettings(algo, requestLimits) // While a 429 holds the server's gate, no new batch goes out: the queue // waits instead of draining into refusals (ocl_issues#2849). const finished = await runWithConcurrency(chunk(_rows, batchSize), { @@ -2374,7 +2375,7 @@ const MapProject = () => { // AutomatchRun's config snapshot records what ran (ocl_online#274). let _selectedAlgos = map( filter(algosSelected, algo => selectedAlgos.includes(algo.id)), - algo => applyRequestSettings(algo, fullRequestLimits) + algo => applyRequestSettings(algo, requestLimits) ) let subActions = [...map(_selectedAlgos, algo => algo.name || algo.id)] subActions.push('reranker') @@ -5644,7 +5645,7 @@ const MapProject = () => { bridgeEnabled={bridgeEnabled} canBridge={canBridge} isCoreUser={isCoreUser} - fullRequestLimits={fullRequestLimits} + requestLimits={requestLimits} canSelectAIModel={canSelectAIModel} canSetAIOutputLocale={canSetAIOutputLocale} canScispacy={canScispacy} diff --git a/src/components/map-projects/MultiAlgoSelector.jsx b/src/components/map-projects/MultiAlgoSelector.jsx index 105e9b35..9ce83c34 100644 --- a/src/components/map-projects/MultiAlgoSelector.jsx +++ b/src/components/map-projects/MultiAlgoSelector.jsx @@ -111,12 +111,11 @@ export default function MultiAlgoSelector({ maxAlgos=5, repo, isCoreUser, - fullRequestLimits=false, + requestLimits=getRequestLimits(null), }) { const { t } = useTranslation() - // Non-core users get at most 10 rows per batch and 5 concurrent requests - // (ocl_online#274). - const requestLimits = getRequestLimits(fullRequestLimits) + // Non-core users get at most 10 rows per batch, and early-access users 5 + // concurrent requests, preview users 2 (ocl_online#274, ocl_issues#2849). const { setAlert } = React.useContext(OperationsContext) const [expanded, setExpanded] = useState(() => new Map()); const [errors, setErrors] = React.useState({}) @@ -186,8 +185,8 @@ export default function MultiAlgoSelector({ const getRequestFieldsSettings = (sel, algo) => getRequestSettings({ batch_size: sel.batch_size ?? algo?.batch_size, concurrent_requests: sel.concurrent_requests ?? algo?.concurrent_requests, - }, fullRequestLimits) - const getLimitHelperText = max => fullRequestLimits ? undefined : t('map_project.request_limit_up_to', {max}) + }, requestLimits) + const getLimitHelperText = max => requestLimits.capped ? t('map_project.request_limit_up_to', {max}) : undefined const getBatchSizeProps = (sel, algo) => ({ value: getRequestFieldsSettings(sel, algo).batchSize, onChange: e => updateSelected(sel.__key, { batch_size: clampInt(e.target.value, 1, requestLimits.batchSize) }), @@ -314,8 +313,8 @@ export default function MultiAlgoSelector({ ...omit(algo, ['getIcon', 'disabled', 'description', 'url']), id: id, name: name, - batch_size: getRequestSettings(algo, fullRequestLimits).batchSize, - concurrent_requests: getRequestSettings(algo, fullRequestLimits).concurrentRequests, + batch_size: getRequestSettings(algo, requestLimits).batchSize, + concurrent_requests: getRequestSettings(algo, requestLimits).concurrentRequests, __key: Math.random(100).toString() }; diff --git a/src/components/map-projects/__tests__/requestLimits.test.js b/src/components/map-projects/__tests__/requestLimits.test.js index 78d7b859..795ee035 100644 --- a/src/components/map-projects/__tests__/requestLimits.test.js +++ b/src/components/map-projects/__tests__/requestLimits.test.js @@ -1,9 +1,12 @@ /** * Auto Match request limits per algorithm (ocl_online#274). * - * Non-core users can't run more than 5 concurrent requests or batches of - * more than 10 rows, whether they type it in the form or a project saved it - * earlier. Core users and staff keep the form's full range. + * Non-core users can't run batches of more than 10 rows, whether they type it + * in the form or a project saved it earlier. Early-access users keep up to 5 + * requests in flight; preview users 2 (ocl_issues#2849), as many previews at + * once could fill the server's matching capacity. A preview project has at + * most 25 rows (3 batches of 10), so 2 costs one more batch's time. Core users + * and staff keep the form's full range. * * Run with: npm test */ @@ -13,6 +16,10 @@ import assert from 'node:assert/strict' import { hasFullRequestLimits, getRequestLimits, getRequestSettings, applyRequestSettings } from '../requestLimits.js' +const FULL = getRequestLimits({ is_staff: true }) +const EARLY_ACCESS = getRequestLimits({ auth_groups: ['early_access'] }) +const PREVIEW = getRequestLimits({ auth_groups: ['preview'] }) + test('hasFullRequestLimits: core users, staff and superusers', () => { assert.equal(hasFullRequestLimits({ auth_groups: ['core_user'] }), true) assert.equal(hasFullRequestLimits({ is_staff: true }), true) @@ -27,52 +34,79 @@ test('hasFullRequestLimits: preview, early-access and signed-out users are cappe assert.equal(hasFullRequestLimits(undefined), false) }) -test('getRequestLimits: the form range for core users and staff, 10 rows × 5 requests for everyone else', () => { - assert.deepEqual(getRequestLimits(true), { batchSize: 1000, concurrentRequests: 50 }) - assert.deepEqual(getRequestLimits(false), { batchSize: 10, concurrentRequests: 5 }) +test('getRequestLimits: the form range for core users and staff', () => { + assert.deepEqual(getRequestLimits({ auth_groups: ['core_user'] }), { batchSize: 1000, concurrentRequests: 50, capped: false }) + assert.deepEqual(FULL, { batchSize: 1000, concurrentRequests: 50, capped: false }) +}) + +test('getRequestLimits: early access, 10 rows × 5 requests', () => { + assert.deepEqual(EARLY_ACCESS, { batchSize: 10, concurrentRequests: 5, capped: true }) +}) + +test('getRequestLimits: preview, 10 rows × 2 requests', () => { + assert.deepEqual(PREVIEW, { batchSize: 10, concurrentRequests: 2, capped: true }) +}) + +test('getRequestLimits: a user in no tier, or signed out, gets the preview limits', () => { + assert.deepEqual(getRequestLimits({ auth_groups: [] }), PREVIEW) + assert.deepEqual(getRequestLimits({ auth_groups: ['mapper_ai_assistant'] }), PREVIEW) + assert.deepEqual(getRequestLimits(null), PREVIEW) +}) + +test('getRequestLimits: the highest tier wins', () => { + assert.deepEqual(getRequestLimits({ auth_groups: ['preview', 'early_access'] }), EARLY_ACCESS) + assert.deepEqual(getRequestLimits({ auth_groups: ['early_access', 'core_user'] }), FULL) }) test('getRequestSettings: a capped user\'s saved project runs within the cap', () => { // the ticket's case: batch 1 × 25 concurrent semantic requests - assert.deepEqual(getRequestSettings({ batch_size: 1, concurrent_requests: 25 }, false), { batchSize: 1, concurrentRequests: 5 }) - assert.deepEqual(getRequestSettings({ batch_size: 1000, concurrent_requests: 50 }, false), { batchSize: 10, concurrentRequests: 5 }) + assert.deepEqual(getRequestSettings({ batch_size: 1, concurrent_requests: 25 }, EARLY_ACCESS), { batchSize: 1, concurrentRequests: 5 }) + assert.deepEqual(getRequestSettings({ batch_size: 1, concurrent_requests: 25 }, PREVIEW), { batchSize: 1, concurrentRequests: 2 }) + assert.deepEqual(getRequestSettings({ batch_size: 1000, concurrent_requests: 50 }, EARLY_ACCESS), { batchSize: 10, concurrentRequests: 5 }) + assert.deepEqual(getRequestSettings({ batch_size: 1000, concurrent_requests: 50 }, PREVIEW), { batchSize: 10, concurrentRequests: 2 }) // ocl-search's shipped default is 50 rows per batch - assert.deepEqual(getRequestSettings({ batch_size: 50, concurrent_requests: 2 }, false), { batchSize: 10, concurrentRequests: 2 }) + assert.deepEqual(getRequestSettings({ batch_size: 50, concurrent_requests: 2 }, EARLY_ACCESS), { batchSize: 10, concurrentRequests: 2 }) }) test('getRequestSettings: values within the cap are kept', () => { - assert.deepEqual(getRequestSettings({ batch_size: 10, concurrent_requests: 2 }, false), { batchSize: 10, concurrentRequests: 2 }) - assert.deepEqual(getRequestSettings({ batch_size: 3, concurrent_requests: 5 }, false), { batchSize: 3, concurrentRequests: 5 }) + assert.deepEqual(getRequestSettings({ batch_size: 10, concurrent_requests: 2 }, EARLY_ACCESS), { batchSize: 10, concurrentRequests: 2 }) + assert.deepEqual(getRequestSettings({ batch_size: 3, concurrent_requests: 5 }, EARLY_ACCESS), { batchSize: 3, concurrentRequests: 5 }) + assert.deepEqual(getRequestSettings({ batch_size: 3, concurrent_requests: 1 }, PREVIEW), { batchSize: 3, concurrentRequests: 1 }) }) test('getRequestSettings: core users and staff run what the project saved, even past the form\'s range', () => { - assert.deepEqual(getRequestSettings({ batch_size: 1, concurrent_requests: 25 }, true), { batchSize: 1, concurrentRequests: 25 }) - assert.deepEqual(getRequestSettings({ batch_size: 50, concurrent_requests: 2 }, true), { batchSize: 50, concurrentRequests: 2 }) - assert.deepEqual(getRequestSettings({ batch_size: 2000, concurrent_requests: 75 }, true), { batchSize: 2000, concurrentRequests: 75 }) + assert.deepEqual(getRequestSettings({ batch_size: 1, concurrent_requests: 25 }, FULL), { batchSize: 1, concurrentRequests: 25 }) + assert.deepEqual(getRequestSettings({ batch_size: 50, concurrent_requests: 2 }, FULL), { batchSize: 50, concurrentRequests: 2 }) + assert.deepEqual(getRequestSettings({ batch_size: 2000, concurrent_requests: 75 }, FULL), { batchSize: 2000, concurrentRequests: 75 }) }) test('getRequestSettings: missing or invalid values fall back to 10 rows × 1 request, as before', () => { - for (const fullLimits of [true, false]) { - assert.deepEqual(getRequestSettings({}, fullLimits), { batchSize: 10, concurrentRequests: 1 }) - assert.deepEqual(getRequestSettings(undefined, fullLimits), { batchSize: 10, concurrentRequests: 1 }) - assert.deepEqual(getRequestSettings({ batch_size: 0, concurrent_requests: 0 }, fullLimits), { batchSize: 10, concurrentRequests: 1 }) - assert.deepEqual(getRequestSettings({ batch_size: 'abc', concurrent_requests: -3 }, fullLimits), { batchSize: 10, concurrentRequests: 1 }) + for (const limits of [FULL, EARLY_ACCESS, PREVIEW]) { + assert.deepEqual(getRequestSettings({}, limits), { batchSize: 10, concurrentRequests: 1 }) + assert.deepEqual(getRequestSettings(undefined, limits), { batchSize: 10, concurrentRequests: 1 }) + assert.deepEqual(getRequestSettings({ batch_size: 0, concurrent_requests: 0 }, limits), { batchSize: 10, concurrentRequests: 1 }) + assert.deepEqual(getRequestSettings({ batch_size: 'abc', concurrent_requests: -3 }, limits), { batchSize: 10, concurrentRequests: 1 }) } }) +test('getRequestSettings: without limits, the preview cap applies', () => { + assert.deepEqual(getRequestSettings({ batch_size: 50, concurrent_requests: 25 }), { batchSize: 10, concurrentRequests: 2 }) +}) + test('getRequestSettings: numeric strings are read as numbers', () => { - assert.deepEqual(getRequestSettings({ batch_size: '8', concurrent_requests: '9' }, false), { batchSize: 8, concurrentRequests: 5 }) + assert.deepEqual(getRequestSettings({ batch_size: '8', concurrent_requests: '9' }, EARLY_ACCESS), { batchSize: 8, concurrentRequests: 5 }) }) test('applyRequestSettings: a capped user\'s algorithm carries the settings the run uses, for the run and its record', () => { const algo = { id: 'ocl-semantic', type: 'ocl-semantic', batch_size: 1, concurrent_requests: 25, query_params: { semantic: true } } - assert.deepEqual(applyRequestSettings(algo, false), { ...algo, batch_size: 1, concurrent_requests: 5 }) - assert.deepEqual(applyRequestSettings({ id: 'ocl-search' }, false), { id: 'ocl-search', batch_size: 10, concurrent_requests: 1 }) + assert.deepEqual(applyRequestSettings(algo, EARLY_ACCESS), { ...algo, batch_size: 1, concurrent_requests: 5 }) + assert.deepEqual(applyRequestSettings(algo, PREVIEW), { ...algo, batch_size: 1, concurrent_requests: 2 }) + assert.deepEqual(applyRequestSettings({ id: 'ocl-search' }, EARLY_ACCESS), { id: 'ocl-search', batch_size: 10, concurrent_requests: 1 }) // the saved project is not changed assert.equal(algo.concurrent_requests, 25) }) test('applyRequestSettings: core users and staff get the algorithm back untouched', () => { const algo = { id: 'ocl-semantic', batch_size: 2000, concurrent_requests: 75 } - assert.equal(applyRequestSettings(algo, true), algo) + assert.equal(applyRequestSettings(algo, FULL), algo) }) diff --git a/src/components/map-projects/requestLimits.js b/src/components/map-projects/requestLimits.js index b447c5da..c3ede851 100644 --- a/src/components/map-projects/requestLimits.js +++ b/src/components/map-projects/requestLimits.js @@ -4,24 +4,35 @@ * * $match is throttled per minute, not per request in flight, so a run at * batch 1 × 25 concurrent requests puts 25 semantic searches on the search - * cluster at once. Everyone but core users and staff gets at most 5 requests - * in flight of at most 10 rows each, in the form and at run time, where the - * cap also holds a project's earlier saved settings. Core users and staff are - * unchanged: the form's range, and whatever the project saved at run time. + * cluster at once. Everyone but core users and staff gets at most 10 rows per + * request, in the form and at run time, where the cap also holds a project's + * earlier saved settings: early-access users up to 5 requests in flight, + * preview users 2 (ocl_issues#2849), since many previews at once could fill + * the server's matching capacity. Core users and staff are unchanged: the + * form's range, and whatever the project saved at run time. */ export const DEFAULT_BATCH_SIZE = 10 export const DEFAULT_CONCURRENT_REQUESTS = 1 -const FORM_LIMITS = { batchSize: 1000, concurrentRequests: 50 } -const CAPPED_LIMITS = { batchSize: 10, concurrentRequests: 5 } +const FULL_LIMITS = { batchSize: 1000, concurrentRequests: 50, capped: false } +const EARLY_ACCESS_LIMITS = { batchSize: 10, concurrentRequests: 5, capped: true } +const PREVIEW_LIMITS = { batchSize: 10, concurrentRequests: 2, capped: true } const inAuthGroup = (user, group) => Boolean(user?.auth_groups?.some(name => name.includes(group))) export const hasFullRequestLimits = user => Boolean(user?.is_staff || user?.is_superuser || inAuthGroup(user, 'core_user')) -// The most rows per request, and requests in flight, a user may type in the form. -export const getRequestLimits = fullLimits => fullLimits ? FORM_LIMITS : CAPPED_LIMITS +// The most rows per request, and requests in flight, a user may type in the +// form, by their highest tier. capped: false for core users and staff, whose +// saved settings run as they are. +export const getRequestLimits = user => { + if(hasFullRequestLimits(user)) + return FULL_LIMITS + if(inAuthGroup(user, 'early_access')) + return EARLY_ACCESS_LIMITS + return PREVIEW_LIMITS +} const toCount = value => { const n = Number.parseInt(value, 10) @@ -29,23 +40,23 @@ const toCount = value => { } // The batch size and concurrency an algorithm runs with: its saved values, or -// the defaults, capped for everyone but core users and staff. -export const getRequestSettings = (algo, fullLimits) => { +// the defaults, within the user's limits (getRequestLimits). +export const getRequestSettings = (algo, limits = PREVIEW_LIMITS) => { const batchSize = toCount(algo?.batch_size) || DEFAULT_BATCH_SIZE const concurrentRequests = toCount(algo?.concurrent_requests) || DEFAULT_CONCURRENT_REQUESTS - if(fullLimits) + if(!limits.capped) return { batchSize, concurrentRequests } return { - batchSize: Math.min(batchSize, CAPPED_LIMITS.batchSize), - concurrentRequests: Math.min(concurrentRequests, CAPPED_LIMITS.concurrentRequests), + batchSize: Math.min(batchSize, limits.batchSize), + concurrentRequests: Math.min(concurrentRequests, limits.concurrentRequests), } } // A copy of a run's algorithm carrying the settings it runs with, so the // AutomatchRun record matches what ran. Core users and staff get it back as is. -export const applyRequestSettings = (algo, fullLimits) => { - if(fullLimits) +export const applyRequestSettings = (algo, limits = PREVIEW_LIMITS) => { + if(!limits.capped) return algo - const { batchSize, concurrentRequests } = getRequestSettings(algo, fullLimits) + const { batchSize, concurrentRequests } = getRequestSettings(algo, limits) return { ...algo, batch_size: batchSize, concurrent_requests: concurrentRequests } } From ec0fb650a32cf5b341f90b0aa4e6fcc44c408f3c Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Tue, 29 Sep 2026 23:18:46 -0400 Subject: [PATCH 03/14] OpenConceptLab/ocl_issues#2849 | Codex pass 1: caps hold under extended pauses, Stop never waits on a hung send, the sweep waits for queued reranks - A gate wait now has a deadline: other requests' 429s extending the pause can't carry a request past its cap. A Retry-After past the cap still pauses the gate first, so the other requests hold back instead of each asking again. - The scheduler doesn't hold its queue for a pause longer than the cap (those batches end throttled at once), and Stop returns even while a batch in flight never answers (untilCancelled). The rerank sweep now runs on the same scheduler; the bridge and ScispaCy loops watch Stop the same way. - Once a request for an algorithm, or a rerank, stays refused for the whole cap, the run stops asking: the rest of its rows end throttled at once, instead of each waiting out another 30 minutes. - The sweep waits for a row's rerank that the debounced path already queued, then proposes the row's mapping; before, it skipped the row, and with the 2-rerank limit most rows were queued by then. - A batch that answers after a newer run started isn't merged into it, and a superseded run's rerank leaves the new run's stages alone. - A throttled rerank shows "Server busy" in the row panel. - The logs POST keeps its payload per project and sends the snapshot taken when a save started, so a retry can't post another project's log. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 143 +++++++++++------- .../map-projects/__tests__/autosave.test.js | 29 ++++ .../map-projects/__tests__/matchBatch.test.js | 43 ++++++ .../__tests__/rowProgress.test.js | 14 ++ src/components/map-projects/autosave.js | 15 +- src/components/map-projects/matchBatch.js | 35 +++-- src/components/map-projects/rowProgress.js | 3 + src/services/__tests__/capacity.test.js | 56 +++++++ src/services/capacity.js | 53 +++++-- 9 files changed, 311 insertions(+), 80 deletions(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index caf8257e..cbb06c0e 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -71,7 +71,7 @@ import { OperationsContext } from '../app/LayoutContext'; import APIService, { isTransientNetworkError, retryWithBackoff } from '../../services/APIService'; import { buildAttributionHeaders, buildConfigSnapshot, summarizeRunCompletion } from '../../services/attribution' -import { CAPACITY_WAIT_CAP_MS, createCapacityGate, createLimiter, isRetryableError, requestWithCapacityRetry, sleepUnlessCancelled } from '../../services/capacity' +import { CAPACITY_WAIT_CAP_MS, createCapacityGate, createLimiter, isRetryableError, requestWithCapacityRetry, sleepUnlessCancelled, untilCancelled } from '../../services/capacity' import { highlightTexts, dropVersion, getCurrentUser, hasAuthGroup, hasCapability, getMapperPreview, getNewProjectBlockReason, downloadObject, currentUserToken, refreshCurrentUserCapabilitiesCache } from '../../common/utils'; import { WHITE, SURFACE_COLORS, TEXT_GRAY } from '../../common/colors'; @@ -589,8 +589,9 @@ const MapProject = () => { // in-flight check"). Replaces the legacy "wait for every algo to // complete" trigger with: any ConceptRow with rerank_score === undefined // makes its row eligible; debounce coalesces rapid algo completions; - // in-flight set prevents double-firing. - const inFlightRerankRef = React.useRef(new Set()) + // in-flight map (row -> a promise that settles when its rerank ends) + // prevents double-firing. + const inFlightRerankRef = React.useRef(new Map()) const rerankDebounceRef = React.useRef({}) const rerankRerunNeededRef = React.useRef(new Set()) // Forward-ref pointer for mergeIntoRowMatchState (declared earlier in @@ -1579,6 +1580,7 @@ const MapProject = () => { }) return } + const rowLogsForSave = options.logs || logsRef.current const projectLogsForSave = options.projectLogs || projectLogsRef.current setIsSaving(true) const f = getFileObjectFromRows() @@ -1677,7 +1679,9 @@ const MapProject = () => { if(!isAutoSave) baseSetAlert({severity: 'success', message: t('map_project.successfully_saved'), duration: 2000}) - postProjectLogs(response.data.url) + // The logs as they were when this save started: it can land after + // the user moved on to another project. + postProjectLogs(response.data.url, {row_logs: rowLogsForSave, project_logs: savedProjectLogs}) } else if(options.leaving && status !== 401) { // The user has left the project, so no dialog can show: say so once. baseSetAlert({severity: 'error', message: t('map_project.leave_save_failed', {name: name || project?.name, detail: errorData?.detail || t('unknown_error')}), duration: 12000}) @@ -1713,27 +1717,23 @@ const MapProject = () => { projectLogsRef.current = newLogs setProjectLogs(newLogs) if(project?.url) - postProjectLogs(project.url) + postProjectLogs(project.url, {row_logs: logsRef.current, project_logs: newLogs}) } - // POSTs the project's logs as they stand when it sends, waiting out a busy - // server. One POST at a time per project: each sends the whole log, so an - // older one that waited out a 429 mustn't land after a newer one - // (ocl_issues#2849). Best effort, like before. - const postProjectLogs = url => { + // POSTs the project's logs, waiting out a busy server. One POST at a time + // per project: each sends the whole log, so an older one that waited out a + // 429 mustn't land after a newer one, and a retry sends the latest log + // triggered for that project, never another project's (ocl_issues#2849). + // Best effort, like before. + const postProjectLogs = (url, logs) => { if(!url) return if(!logsSendersRef.current.has(url)) - logsSendersRef.current.set(url, createLatestSender(() => requestWithCapacityRetry( - () => APIService.new().overrideURL(url).appendToUrl('logs/').request( - 'POST', - {logs: {row_logs: logsRef.current, project_logs: projectLogsRef.current}}, - null, - {handlesThrottle: true} - ), + logsSendersRef.current.set(url, createLatestSender(getPayload => requestWithCapacityRetry( + () => APIService.new().overrideURL(url).appendToUrl('logs/').request('POST', getPayload(), null, {handlesThrottle: true}), {maxWaitMs: INTERACTIVE_WAIT_CAP_MS} ))) - logsSendersRef.current.get(url).trigger() + logsSendersRef.current.get(url).trigger({logs}) } const scheduleAutoSave = reason => autosaveSchedulerRef.current.schedule(reason) @@ -1805,6 +1805,16 @@ const MapProject = () => { } } + // Once a request for an algorithm (or rerank) stayed refused for the whole + // cap, the run stops asking for it: the rest of its rows end throttled at + // once, "not run, retry", instead of each waiting out the cap again. + const isRunThrottled = algoId => Boolean(throttledRunRowsRef.current[algoId]?.length) + const markRowsThrottled = (indexes, algoId, logExtras = {}) => indexes.forEach(index => { + markAlgo(index, algoId, -4) + log({action: algoId === 'rerank' ? 'rerank_throttled' : 'algo_throttled', description: t('map_project.row_throttled'), extras: {...logExtras, skipped: true}}, index) + noteThrottledRunRow(algoId, index) + }) + // ── ocl_online#105 Phase 5: AutomatchRun attribution ────────────────────── // Build the X-OCL-Request-Source + X-OCL-Event-Metadata headers for a single // per-row backend call. Attribution is by explicit ORIGIN, not ambient: only @@ -2302,12 +2312,23 @@ const MapProject = () => { const { batchSize, concurrentRequests } = getRequestSettings(algo, requestLimits) // While a 429 holds the server's gate, no new batch goes out: the queue // waits instead of draining into refusals (ocl_issues#2849). + const isQuotaStopped = () => Boolean(matchQuotaStopRef.current && spendsMatchQuota(algo, {canBridge})) const finished = await runWithConcurrency(chunk(_rows, batchSize), { concurrency: concurrentRequests, gate: getCapacityGate(getMatchAPIService(algo).URL), + maxHoldMs: CAPACITY_WAIT_CAP_MS, shouldAbort: () => abortRef.current, - shouldSkipRest: () => Boolean(matchQuotaStopRef.current && spendsMatchQuota(algo, {canBridge})), + shouldSkipRest: () => isQuotaStopped() || isRunThrottled(algo.id), + // A quota stop leaves the rest not run, as before; a server that + // stayed too busy leaves them throttled. + onSkipped: rowBatches => { + if(!isQuotaStopped()) + markRowsThrottled(map(flatten(rowBatches), '__index'), algo.id, getAlgoLogExtras(algo)) + }, run: rowBatch => processBatch(_repo, rowBatch, algo).then((data) => { + // A batch that answers after a newer run started belongs to no run. + if(!isCurrentRun()) + return // Populate rowMatchState before any consumer (setStateViews / // setAutoMatched) tries to read it. The per-row fetch flow // routes through `onResponse` which already calls @@ -2345,31 +2366,17 @@ const MapProject = () => { setLoadingMatches(false) }; + // Reranks the run's rows, at most maxConcurrent at once, through the same + // limiter as every other rerank. Stop returns at once. const processRerankWithConcurrency = async (_rows, maxConcurrent = 2) => { - const queue = _rows.slice(); - const activeRequests = new Set(); - - while (queue.length > 0 || activeRequests.size > 0) { - while (queue.length > 0 && activeRequests.size < maxConcurrent) { - if (abortRef.current) { - setLoadingMatches(false) - return - } - if(rerankQuotaStopRef.current) { - queue.length = 0 - break - } - - const row = queue.shift(); - const promise = rerank(row.__index, true) - .catch(() => null) - .finally(() => activeRequests.delete(promise)); - activeRequests.add(promise); - } - - if(activeRequests.size > 0) - await Promise.race(activeRequests); - } + const finished = await runWithConcurrency(_rows, { + concurrency: maxConcurrent, + shouldAbort: () => abortRef.current, + shouldSkipRest: () => Boolean(rerankQuotaStopRef.current), + run: row => rerank(row.__index, true).catch(() => null), + }) + if(!finished) + setLoadingMatches(false) }; // Capped users' algorithms carry the settings they run with, so the // AutomatchRun's config snapshot records what ran (ocl_online#274). @@ -2597,9 +2604,14 @@ const MapProject = () => { break; }; if (matchQuotaStopRef.current) break; + if(isRunThrottled(algo.id)) { + markRowsThrottled(map(_rows.slice(index), '__index'), algo.id, getAlgoLogExtras(algo)) + break + } markAlgo(_rows[index].__index, algo.id, 0) - await fetchBridgeCandidates(_rows[index], 0, undefined, undefined, undefined, false, true, ((response, payload) => { + // Stop doesn't wait on a request that never answers (ocl_issues#2849). + await untilCancelled(fetchBridgeCandidates(_rows[index], 0, undefined, undefined, undefined, false, true, ((response, payload) => { const index = payload.rows[0].__index const results = (isArray(response) ? response : response?.data) log({action: 'algo_finished', extras: getAlgoLogExtras(algo)}, index) @@ -2617,7 +2629,7 @@ const MapProject = () => { rawResponse: response }), {isRunTraffic: true}) } - })); // wait for completion + })), () => abortRef.current); // wait for completion await new Promise(resolve => setTimeout(resolve, 200)); // 1s delay } const now = moment() @@ -2633,10 +2645,15 @@ const MapProject = () => { break; }; + if(isRunThrottled(algo.id) || isRunThrottled('ocl-scispacy-loinc')) { + markRowsThrottled(map(_rows.slice(index), '__index'), algo.id, getAlgoLogExtras(algo)) + break + } markAlgo(_rows[index].__index, algo.id, 0) setLoadingMatches(true) - await fetchScispacyCandidates(_rows[index], false, false, true, (response => { + // Stop doesn't wait on a request that never answers (ocl_issues#2849). + await untilCancelled(fetchScispacyCandidates(_rows[index], false, false, true, (response => { const _index = _rows[index].__index const results = [{row: _rows[index], results: fromScispacyResultsToConcepts(getScispacyRowResults(response.data, _index))}] log({action: 'algo_finished', extras: getAlgoLogExtras(algo)}, _index) @@ -2654,7 +2671,7 @@ const MapProject = () => { rawResponse: response }), {isRunTraffic: true}) } - })); // wait for completion + })), () => abortRef.current); // wait for completion await new Promise(resolve => setTimeout(resolve, 500)); // 1s delay } const now = moment() @@ -4132,9 +4149,16 @@ const MapProject = () => { const rerank = async (_index, isBulk=false, isRunTraffic=isBulk) => { const index = isNumber(_index) ? _index : rowIndex if(!isNumber(index)) return null - if(inFlightRerankRef.current.has(index)) { - // Another rerank is in flight for this row; flag a rerun and bail. + const rerankInFlight = inFlightRerankRef.current.get(index) + if(rerankInFlight) { + // Another rerank is in flight, or waiting its turn, for this row; flag + // a rerun. The run's sweep waits for it, then proposes the row's + // mapping, which only the sweep's own call does (ocl_issues#2849). rerankRerunNeededRef.current.add(index) + if(!isBulk) + return null + await rerankInFlight + setTimeout(() => setAutoMatched([index]), 1000) return null } // Wait for any in-flight $lookups for this row's concepts. buildRerankRowsForRow @@ -4171,11 +4195,21 @@ const MapProject = () => { setTimeout(() => setAutoMatched([index]), 1000) return null } - inFlightRerankRef.current.add(index) + // A rerank in this run already stayed refused for the whole cap: don't + // ask again, the row's rerank is throttled ("not run, retry"). + if(isRunTraffic && isRunThrottled('rerank')) { + markRowsThrottled([index], 'rerank', {algo: 'reranker'}) + return null + } + let settleInFlight + inFlightRerankRef.current.set(index, new Promise(resolve => { settleInFlight = resolve })) markAlgo(index, 'rerank', 0) const service = APIService.concepts().appendToUrl('$rerank/') // A run's rerank ends when the run stops; a row reranked by hand doesn't. const isCancelled = isRunTraffic ? getRunStopCheck() : () => false + // A run's rerank that ends after a newer run started leaves its stages alone. + const runTicket = autoMatchRunTicketRef.current + const isSuperseded = () => isRunTraffic && runTicket !== autoMatchRunTicketRef.current let release = null try { // At most RERANK_MAX_IN_FLIGHT $rerank calls at once, whatever fired @@ -4184,7 +4218,8 @@ const MapProject = () => { if(!release) { // Stopped before its turn; a stopped run reranks nothing more. rerankRerunNeededRef.current.delete(index) - markAlgo(index, 'rerank', -1) + if(!isSuperseded()) + markAlgo(index, 'rerank', -1) return null } // Score the candidates that arrived while this call waited its turn too. @@ -4203,11 +4238,14 @@ const MapProject = () => { }) if(result.reason === 'cancelled') { rerankRerunNeededRef.current.delete(index) - markAlgo(index, 'rerank', -1) + if(!isSuperseded()) + markAlgo(index, 'rerank', -1) return null } if(result.reason === 'throttled') { log({action: 'rerank_throttled', description: t('map_project.row_throttled'), extras: {attempts: result.attempts, waited_ms: result.waitedMs}}, index) + if(isSuperseded()) + return null markAlgo(index, 'rerank', -4) if(isRunTraffic) noteThrottledRunRow('rerank', index) @@ -4270,6 +4308,7 @@ const MapProject = () => { } finally { release?.() inFlightRerankRef.current.delete(index) + settleInFlight() // If new ConceptRows arrived while we were in flight, fire again. if(rerankRerunNeededRef.current.has(index)) { rerankRerunNeededRef.current.delete(index) diff --git a/src/components/map-projects/__tests__/autosave.test.js b/src/components/map-projects/__tests__/autosave.test.js index 01ba79f0..181c1809 100644 --- a/src/components/map-projects/__tests__/autosave.test.js +++ b/src/components/map-projects/__tests__/autosave.test.js @@ -364,3 +364,32 @@ test('createLatestSender: after it settles, the next trigger sends again', async await sender.trigger() assert.equal(sends, 2) }) + +// Codex review, pass 1: the payload belongs to the sender (one per project), +// so a retry for project A can't pick up project B's log from shared state. +test('createLatestSender: the send reads the latest payload triggered on this sender, on every attempt', async () => { + const gate = deferred() + const seen = [] + const sender = createLatestSender(async getPayload => { + seen.push(getPayload()) + if(seen.length === 1) { + await gate.promise + // a retry within the same send reads it again + seen.push(getPayload()) + } + }) + const first = sender.trigger({logs: 'a1'}) + sender.trigger({logs: 'a2'}) + sender.trigger({logs: 'a3'}) + gate.resolve() + await first + assert.deepEqual(seen, [{logs: 'a1'}, {logs: 'a3'}, {logs: 'a3'}]) +}) + +test('createLatestSender: two senders keep their own payloads', async () => { + const seen = [] + const a = createLatestSender(async getPayload => { seen.push(['a', getPayload()]) }) + const b = createLatestSender(async getPayload => { seen.push(['b', getPayload()]) }) + await Promise.all([a.trigger('project A'), b.trigger('project B')]) + assert.deepEqual(seen.sort(), [['a', 'project A'], ['b', 'project B']]) +}) diff --git a/src/components/map-projects/__tests__/matchBatch.test.js b/src/components/map-projects/__tests__/matchBatch.test.js index 7692c0f9..14ad45f1 100644 --- a/src/components/map-projects/__tests__/matchBatch.test.js +++ b/src/components/map-projects/__tests__/matchBatch.test.js @@ -495,3 +495,46 @@ test('runWithConcurrency: shouldSkipRest drops the queue but lets the batches in assert.equal(done, true) assert.deepEqual(finished.sort(), [1, 2]) }) + +test('runWithConcurrency: onSkipped gets the batches shouldSkipRest dropped, once', async () => { + let skip = false + const skipped = [] + await runWithConcurrency([1, 2, 3, 4, 5], { + concurrency: 1, + shouldSkipRest: () => skip, + onSkipped: items => skipped.push(items), + run: async item => { if(item === 2) skip = true }, + }) + assert.deepEqual(skipped, [[3, 4, 5]]) +}) + +test('runWithConcurrency: onSkipped isn\'t called when nothing was dropped', async () => { + const skipped = [] + await runWithConcurrency([1, 2], {concurrency: 2, onSkipped: items => skipped.push(items), run: async () => {}}) + assert.deepEqual(skipped, []) +}) + +// Codex review, pass 1 +test('runWithConcurrency: Stop returns even while a batch in flight never answers', async () => { + let stopped = false + setTimeout(() => { stopped = true }, 20) + const started = Date.now() + const done = await runWithConcurrency([1, 2, 3], { + concurrency: 1, + pollMs: 5, + shouldAbort: () => stopped, + run: () => new Promise(() => {}), + }) + assert.equal(done, false) + assert.ok(Date.now() - started < 1000) +}) + +test('runWithConcurrency: a pause longer than maxHoldMs doesn\'t hold the queue (its batches end throttled at once instead)', async () => { + const gate = createCapacityGate() + gate.pause(60 * 60 * 1000) + const ran = [] + const started = Date.now() + await runWithConcurrency([1, 2, 3], {concurrency: 1, gate, maxHoldMs: 30 * 60 * 1000, pollMs: 5, run: async item => { ran.push(item) }}) + assert.deepEqual(ran, [1, 2, 3]) + assert.ok(Date.now() - started < 1000) +}) diff --git a/src/components/map-projects/__tests__/rowProgress.test.js b/src/components/map-projects/__tests__/rowProgress.test.js index 48bb7cfc..a94a37c2 100644 --- a/src/components/map-projects/__tests__/rowProgress.test.js +++ b/src/components/map-projects/__tests__/rowProgress.test.js @@ -57,3 +57,17 @@ test('getRowProgressLabel: an algorithm that can\'t run for this user (-3, n/a) assert.equal(getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': -3}, ALGOS, {t}).label, undefined) assert.equal(getRowProgressLabel({'ocl-semantic': -2, 'ocl-bridge': -3}, ALGOS, {t}).status, 'partial') }) + +// Codex review, pass 1: the rerank's own stage counts too. +test('getRowProgressLabel: a throttled rerank asks for a retry once the algorithms are done', () => { + assert.deepEqual( + getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': 1, rerank: -4}, ALGOS, {t}), + {label: 'map_project.row_throttled', status: 'throttled'}, + ) + assert.equal(getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': -3, rerank: -4}, ALGOS, {t}).status, 'throttled') +}) + +test('getRowProgressLabel: a done or failed rerank leaves the label as before', () => { + assert.equal(getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': 1, rerank: 1}, ALGOS, {t}).label, undefined) + assert.equal(getRowProgressLabel({'ocl-semantic': 1, 'ocl-bridge': 1, rerank: -2}, ALGOS, {t}).label, undefined) +}) diff --git a/src/components/map-projects/autosave.js b/src/components/map-projects/autosave.js index 067c57b8..1b32b6e8 100644 --- a/src/components/map-projects/autosave.js +++ b/src/components/map-projects/autosave.js @@ -144,19 +144,23 @@ export const installUnloadGuard = () => { /** * One send at a time; triggers that arrive during it coalesce into one more - * send, which reads its payload then (ocl_issues#2849). For the project logs - * POST, which sends the whole log: once it waits out a 429, an older POST - * could otherwise land after a newer one and overwrite it. trigger() resolves, + * send (ocl_issues#2849). For the project logs POST, which sends the whole + * log: once it waits out a 429, an older POST could otherwise land after a + * newer one and overwrite it. send(getPayload) reads the latest payload + * triggered on this sender, on each attempt; keep one sender per project, so + * a retry for one project never sends another's. trigger(payload) resolves, * never rejects, once nothing is left to send. */ export const createLatestSender = send => { let chain = null let pending = false + let latest + const getPayload = () => latest const loop = async () => { do { pending = false try { - await send() + await send(getPayload) } catch (_) { // Best effort: the next change sends the whole log again. } @@ -164,7 +168,8 @@ export const createLatestSender = send => { chain = null } return { - trigger: () => { + trigger: payload => { + latest = payload if(chain) pending = true else diff --git a/src/components/map-projects/matchBatch.js b/src/components/map-projects/matchBatch.js index 8d143439..69d274d7 100644 --- a/src/components/map-projects/matchBatch.js +++ b/src/components/map-projects/matchBatch.js @@ -11,7 +11,7 @@ * busy server makes a run slower, not broken. */ import { isPreviewLimitError } from './previewLimits.js' -import { isRetryableError, requestWithCapacityRetry } from '../../services/capacity.js' +import { isRetryableError, requestWithCapacityRetry, untilCancelled } from '../../services/capacity.js' // How long the row panel waits out a 429 before showing the row as throttled. // A person is watching it, so less than a run's 30 minutes. @@ -103,37 +103,46 @@ export const runMatchBatch = async ({ * The bulk Auto Match scheduler: runs each item through run(item) with at most * concurrency in flight. While the gate is paused (a request got a 429) it * sends nothing new, and waits for the gate to reopen or a request in flight - * to settle, so a throttled run doesn't drain its queue into more refusals. + * to settle, so a throttled run doesn't drain its queue into more refusals. A + * pause longer than maxHoldMs doesn't hold the queue: its items go out and + * end throttled at once, without sending. * - * shouldAbort stops it at once (the requests in flight are left to finish on - * their own) and returns false. shouldSkipRest drops the rest of the queue but - * waits for the requests in flight. Otherwise returns true once every item ran. + * shouldAbort stops it at once, even while a request in flight never answers + * (those are left to finish on their own), and returns false. shouldSkipRest + * drops the rest of the queue, handing it to onSkipped, but waits for the + * requests in flight. Otherwise returns true once every item ran. */ export const runWithConcurrency = async (items, { - concurrency, run, shouldAbort = () => false, shouldSkipRest = () => false, gate = null, pollMs, + concurrency, run, shouldAbort = () => false, shouldSkipRest = () => false, onSkipped, gate = null, maxHoldMs = Infinity, pollMs, }) => { const queue = items.slice() const active = new Set() + const isHeld = () => Boolean(gate?.isPaused() && gate.pausedForMs() <= maxHoldMs) while(queue.length || active.size) { while(queue.length && active.size < concurrency) { if(shouldAbort()) return false if(shouldSkipRest()) { - queue.length = 0 + // Empty the queue before the optional call: onSkipped?.() skips + // evaluating its argument when there's no callback. + const skipped = queue.splice(0) + onSkipped?.(skipped) break } - if(gate?.isPaused()) + if(isHeld()) break const item = queue.shift() const promise = run(item).finally(() => active.delete(promise)) active.add(promise) } - if(queue.length && active.size < concurrency && gate?.isPaused()) { - await Promise.race([gate.wait(shouldAbort, {pollMs}), ...active]) + const waitingOn = [...active] + if(queue.length && active.size < concurrency && isHeld()) + waitingOn.push(gate.wait(shouldAbort, {pollMs, maxMs: maxHoldMs})) + if(!waitingOn.length) continue - } - if(active.size) - await Promise.race(active) + const { settled } = await untilCancelled(Promise.race(waitingOn), shouldAbort, {pollMs}) + if(!settled) + return false } return true } diff --git a/src/components/map-projects/rowProgress.js b/src/components/map-projects/rowProgress.js index 924b6c09..3f9d930a 100644 --- a/src/components/map-projects/rowProgress.js +++ b/src/components/map-projects/rowProgress.js @@ -32,6 +32,9 @@ export const getRowProgressLabel = (stageMap, algos, { t, capacityWait = false } // -3: the algorithm isn't available to this user, so there's nothing to wait for. if (stages.every(v => v === 1 || v === -3)) { + // The candidates are in, but the server stayed too busy to rank them. + if(stageMap.rerank === -4) + return {label: t('map_project.row_throttled'), status: 'throttled'} return true } diff --git a/src/services/__tests__/capacity.test.js b/src/services/__tests__/capacity.test.js index c3df31a2..841d7da1 100644 --- a/src/services/__tests__/capacity.test.js +++ b/src/services/__tests__/capacity.test.js @@ -23,6 +23,7 @@ import { getRetryAfterMs, requestWithCapacityRetry, sleepUnlessCancelled, + untilCancelled, } from '../capacity.js' const networkError = () => Object.assign(new Error('Network Error'), {isAxiosError: true, code: 'ERR_NETWORK'}) @@ -483,3 +484,58 @@ test('requestWithCapacityRetry: a throttled wait carries the 429\'s capacity hea assert.deepEqual(waits, [{decision: 'refused', tier: 'preview'}, null]) }) + +// ── Codex review, pass 1 ──────────────────────────────────────────────────── + +test('createCapacityGate: wait gives up once maxMs has passed, even as the pause keeps extending', async () => { + const clock = virtualClock() + const gate = createCapacityGate({now: clock.now}) + gate.pause(10000) + // another request's 429 keeps pushing the pause out + const sleep = async ms => { await clock.sleep(ms); gate.pause(10000) } + assert.equal(await gate.wait(() => false, {sleep, maxMs: 30000}), 'timeout') + assert.ok(clock.t >= 30000 && clock.t <= 30250, `waited ${clock.t}`) +}) + +test('requestWithCapacityRetry: a gate pause that keeps extending ends "throttled" at the cap, without sending', async () => { + const clock = virtualClock() + const gate = createCapacityGate({now: clock.now}) + gate.pause(20000) + const sleep = async ms => { await clock.sleep(ms); gate.pause(20000) } + const {send, sent} = scripted(ok()) + + const result = await requestWithCapacityRetry(send, {gate, now: clock.now, sleep, maxWaitMs: 60000}) + + assert.equal(result.reason, 'throttled') + assert.equal(sent.length, 0) + assert.ok(clock.t <= 60250, `waited ${clock.t}`) +}) + +test('requestWithCapacityRetry: a Retry-After past the cap still pauses the gate, so the other requests hold back too', async () => { + const clock = virtualClock() + const gate = createCapacityGate({now: clock.now}) + + const first = await requestWithCapacityRetry(scripted(throttled(86400)).send, {gate, now: clock.now, sleep: clock.sleep}) + const second = scripted(ok()) + const next = await requestWithCapacityRetry(second.send, {gate, now: clock.now, sleep: clock.sleep}) + + assert.equal(first.reason, 'throttled') + assert.equal(gate.pausedForMs(), 86400 * 1000) + // the next request doesn't ask again: the server said when to come back + assert.equal(next.reason, 'throttled') + assert.equal(second.sent.length, 0) + assert.equal(clock.t, 0) +}) + +test('untilCancelled: the promise\'s value when it settles, or cancelled when Stop comes first', async () => { + assert.deepEqual(await untilCancelled(Promise.resolve(7), () => false, {pollMs: 5}), {settled: true, value: 7}) + let stopped = false + setTimeout(() => { stopped = true }, 15) + const started = Date.now() + assert.deepEqual(await untilCancelled(new Promise(() => {}), () => stopped, {pollMs: 5}), {settled: false}) + assert.ok(Date.now() - started < 500) +}) + +test('untilCancelled: a rejection passes through', async () => { + await assert.rejects(untilCancelled(Promise.reject(new Error('boom')), () => false, {pollMs: 5}), /boom/) +}) diff --git a/src/services/capacity.js b/src/services/capacity.js index 387c3bee..0953cc87 100644 --- a/src/services/capacity.js +++ b/src/services/capacity.js @@ -127,10 +127,16 @@ export const createCapacityGate = ({ now = Date.now } = {}) => { pause: ms => { resumeAt = Math.max(resumeAt, now() + ms) }, pausedForMs, isPaused: () => pausedForMs() > 0, - // Resolves true once the gate is open, false if cancelled first. - wait: async (isCancelled = () => false, { sleep, pollMs } = {}) => { + // Resolves true once the gate is open, false if cancelled first, or + // 'timeout' once maxMs has passed with the gate still paused (other + // requests' 429s can keep extending the pause). + wait: async (isCancelled = () => false, { sleep, pollMs, maxMs = Infinity } = {}) => { + const startedAt = now() while(pausedForMs() > 0) { - if(!(await sleepUnlessCancelled(pausedForMs(), isCancelled, { sleep, pollMs }))) + const leftMs = maxMs - (now() - startedAt) + if(leftMs <= 0) + return 'timeout' + if(!(await sleepUnlessCancelled(Math.min(pausedForMs(), leftMs), isCancelled, { sleep, pollMs }))) return false } return true @@ -198,18 +204,23 @@ export const requestWithCapacityRetry = async (send, { if(gate?.isPaused()) { const pauseMs = gate.pausedForMs() - if(waitedMs + pauseMs > maxWaitMs) + const budgetMs = maxWaitMs - waitedMs + if(pauseMs > budgetMs) return end({ ok: false, reason: 'throttled', error: lastError }) onWait?.({ reason: 'paused', delayMs: pauseMs, status: null, retryAfterMs: pauseMs }) const startedAt = now() - let open = await gate.wait(isCancelled, { sleep, pollMs }) - // Spread out the requests the pause held back. - if(open) - open = await sleepUnlessCancelled(pauseMs * THROTTLE_JITTER * random(), isCancelled, { sleep, pollMs }) + let outcome = await gate.wait(isCancelled, { sleep, pollMs, maxMs: budgetMs }) + // Spread out the requests the pause held back, within the budget. + if(outcome === true) { + const jitterMs = Math.min(pauseMs * THROTTLE_JITTER * random(), Math.max(budgetMs - (now() - startedAt), 0)) + outcome = await sleepUnlessCancelled(jitterMs, isCancelled, { sleep, pollMs }) + } waitedMs += now() - startedAt onWaitEnd?.() - if(!open) + if(outcome === false) return cancelled() + if(outcome === 'timeout' || waitedMs > maxWaitMs) + return end({ ok: false, reason: 'throttled', error: lastError }) continue } @@ -227,9 +238,11 @@ export const requestWithCapacityRetry = async (send, { Math.max(retryAfterMs, MIN_THROTTLE_WAIT_MS) throttles += 1 const delayMs = baseMs * (1 + random() * THROTTLE_JITTER) + // The other requests to this server hold back as long, even when + // this one gives up now. + gate?.pause(baseMs) if(waitedMs + delayMs > maxWaitMs) return end({ ok: false, reason: 'throttled', error: err }) - gate?.pause(baseMs) if(!(await waitFor(delayMs, { reason: 'throttled', status: 429, retryAfterMs, capacity: getCapacityHeaders(err.response) }))) return cancelled() continue @@ -251,6 +264,26 @@ export const requestWithCapacityRetry = async (send, { } } +/** + * Waits for promise, unless isCancelled turns true first: resolves + * {settled: true, value}, or {settled: false} on Stop, leaving the promise to + * finish on its own. A rejection passes through. For a run that must not wait + * on a request that never answers once the user stops it. + */ +export const untilCancelled = (promise, isCancelled = () => false, { pollMs = POLL_MS, sleep = defaultSleep } = {}) => { + let done = false + const outcome = Promise.resolve(promise).then(value => ({ settled: true, value })).finally(() => { done = true }) + const stopped = (async () => { + while(!done) { + if(isCancelled()) + return { settled: false } + await sleep(pollMs) + } + return null + })() + return Promise.race([outcome, stopped.then(result => result || outcome)]) +} + /** * At most max callers hold a slot at once; the rest queue in order. acquire * resolves a release function, or null when isCancelled turned true before From bd81b0f4728b4d1ffda3b5b8cf7daab9912604a6 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Tue, 29 Sep 2026 23:30:24 -0400 Subject: [PATCH 04/14] OpenConceptLab/ocl_issues#2849 | Codex pass 2 and the local e2e: transport timeouts, one refused algorithm no longer blocks the rest, the chain goes on after a failure - Transport timeouts bound a send that never answers: 11 minutes for heavy calls ($match, $rerank, bridge, ScispaCy), longer than the API itself runs a request, so a timeout never cuts short work the server is still doing; 60 s for logs, run records, lookups and search. A hung rerank can't hold a limiter slot, or a hung logs POST the logs, for good. - Stop ends the wait for the AutomatchRun create call too; a run the server opens after that is closed as cancelled. - A Retry-After past the cap no longer pauses the whole server's gate: the e2e check showed one refused algorithm blocking another, the reranks and later runs in the tab for as long. That request gives up alone, and the run stops asking for its algorithm (the rule from pass 1). - The scheduler's hold on a paused gate is 30 minutes in all, however often other requests extend the pause. - The sweep reruns a row after its queued rerank ends, so candidates that arrived meanwhile are scored before the row's mapping is proposed; and a stopped run proposes nothing more. - A superseded run's bridge, ScispaCy and rerank outcomes leave the new run's stages alone. - A save keeps the entries logged while it was in flight, instead of resetting the log to what it held when the save started; a save that lands after another project was opened doesn't touch that project's log. - When ScispaCy or the bridge fails (or stays throttled) for a row matched by hand, the row goes on to its next algorithm and its rerank, as other algorithms do. - The toolbar notice is short in the split view, with the full text in a tooltip. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 189 +++++++++++++----- .../map-projects/__tests__/matchBatch.test.js | 14 ++ src/components/map-projects/matchBatch.js | 18 +- src/i18n/locales/en/translations.json | 1 + src/i18n/locales/es/translations.json | 1 + src/i18n/locales/zh/translations.json | 1 + src/services/__tests__/capacity.test.js | 35 +++- src/services/capacity.js | 16 +- 8 files changed, 204 insertions(+), 71 deletions(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index cbb06c0e..1910f1b4 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -71,7 +71,7 @@ import { OperationsContext } from '../app/LayoutContext'; import APIService, { isTransientNetworkError, retryWithBackoff } from '../../services/APIService'; import { buildAttributionHeaders, buildConfigSnapshot, summarizeRunCompletion } from '../../services/attribution' -import { CAPACITY_WAIT_CAP_MS, createCapacityGate, createLimiter, isRetryableError, requestWithCapacityRetry, sleepUnlessCancelled, untilCancelled } from '../../services/capacity' +import { CAPACITY_WAIT_CAP_MS, HEAVY_REQUEST_TIMEOUT_MS, LIGHT_REQUEST_TIMEOUT_MS, createCapacityGate, createLimiter, isRetryableError, requestWithCapacityRetry, sleepUnlessCancelled, untilCancelled } from '../../services/capacity' import { highlightTexts, dropVersion, getCurrentUser, hasAuthGroup, hasCapability, getMapperPreview, getNewProjectBlockReason, downloadObject, currentUserToken, refreshCurrentUserCapabilitiesCache } from '../../common/utils'; import { WHITE, SURFACE_COLORS, TEXT_GRAY } from '../../common/colors'; @@ -1662,9 +1662,17 @@ const MapProject = () => { user: user.username || user.id, extras: isAutoSave ? {reasons: options.reasons || []} : (isUpdate ? undefined : {project: response.data}) } - const savedProjectLogs = [saveLog, ...projectLogsForSave] - projectLogsRef.current = savedProjectLogs - setProjectLogs(savedProjectLogs) + // Entries logged while the save was in flight are kept. A save that + // lands after the user opened another project uses the logs as they + // were when it started, and leaves the open project's alone + // (ocl_issues#2849). + const isOtherProjectOpen = isUpdate && String(projectIdRef.current) !== String(response.data.id) + const savedProjectLogs = [saveLog, ...(isOtherProjectOpen ? projectLogsForSave : projectLogsRef.current)] + const savedRowLogs = isOtherProjectOpen ? rowLogsForSave : logsRef.current + if(!isOtherProjectOpen) { + projectLogsRef.current = savedProjectLogs + setProjectLogs(savedProjectLogs) + } if(options.closeConfigure !== false) { configSnapshotOnOpenRef.current = null setConfigure(false) @@ -1679,9 +1687,7 @@ const MapProject = () => { if(!isAutoSave) baseSetAlert({severity: 'success', message: t('map_project.successfully_saved'), duration: 2000}) - // The logs as they were when this save started: it can land after - // the user moved on to another project. - postProjectLogs(response.data.url, {row_logs: rowLogsForSave, project_logs: savedProjectLogs}) + postProjectLogs(response.data.url, {row_logs: savedRowLogs, project_logs: savedProjectLogs}) } else if(options.leaving && status !== 401) { // The user has left the project, so no dialog can show: say so once. baseSetAlert({severity: 'error', message: t('map_project.leave_save_failed', {name: name || project?.name, detail: errorData?.detail || t('unknown_error')}), duration: 12000}) @@ -1730,7 +1736,7 @@ const MapProject = () => { return if(!logsSendersRef.current.has(url)) logsSendersRef.current.set(url, createLatestSender(getPayload => requestWithCapacityRetry( - () => APIService.new().overrideURL(url).appendToUrl('logs/').request('POST', getPayload(), null, {handlesThrottle: true}), + () => APIService.new().overrideURL(url).appendToUrl('logs/').request('POST', getPayload(), null, {handlesThrottle: true, timeout: LIGHT_REQUEST_TIMEOUT_MS}), {maxWaitMs: INTERACTIVE_WAIT_CAP_MS} ))) logsSendersRef.current.get(url).trigger({logs}) @@ -1857,13 +1863,28 @@ const MapProject = () => { }), } try { - // A busy server's 429 is waited out, and Stop ends the wait. Nothing - // else is retried: after a 502 the run may exist already, and a retry - // would open a second one (ocl_issues#2849). - const result = await requestWithCapacityRetry( - () => APIService.new().overrideURL(project.url).appendToUrl('auto-match-runs/').request('POST', body, null, {handlesThrottle: true}), - {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, isCancelled: getRunStopCheck(), isRetryable: () => false, onCapacity: noteCapacity} + // A busy server's 429 is waited out, and Stop ends the wait, and the + // wait for an answer. Nothing else is retried: after a 502 or a timeout + // the run may exist already, and a retry would open a second one + // (ocl_issues#2849). + const isStopped = getRunStopCheck() + const request = requestWithCapacityRetry( + () => APIService.new().overrideURL(project.url).appendToUrl('auto-match-runs/').request('POST', body, null, {handlesThrottle: true, timeout: LIGHT_REQUEST_TIMEOUT_MS}), + {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, isCancelled: isStopped, isRetryable: () => false, onCapacity: noteCapacity} ) + const outcome = await untilCancelled(request, isStopped) + if(!outcome.settled) { + // Stopped before the server answered: close a run it opens anyway. + request.then(late => { + const lateId = late.ok && late.response?.data?.id + if(lateId) + APIService.new().overrideURL('/auto-match-runs/' + lateId + '/') + .request('PATCH', {completed_rows: 0, failed_rows: 0, completion_status: 'cancelled'}, null, {handlesThrottle: true, timeout: LIGHT_REQUEST_TIMEOUT_MS}) + .catch(() => {}) + }) + return + } + const result = outcome.value const response = result.ok ? result.response : result.error?.response const status = response?.status const data = response?.data @@ -1905,7 +1926,7 @@ const MapProject = () => { // retried; Stop doesn't end it, so a stopped run is closed too // (ocl_issues#2849). requestWithCapacityRetry never rejects. await requestWithCapacityRetry( - () => APIService.new().overrideURL('/auto-match-runs/' + run.id + '/').request('PATCH', payload, null, {handlesThrottle: true}), + () => APIService.new().overrideURL('/auto-match-runs/' + run.id + '/').request('PATCH', payload, null, {handlesThrottle: true, timeout: LIGHT_REQUEST_TIMEOUT_MS}), {maxWaitMs: INTERACTIVE_WAIT_CAP_MS} ) } @@ -2064,7 +2085,7 @@ const MapProject = () => { service.post = (body, token, headers, query) => { const context = bridgeCallContextRef.current || {} context.posted = true - return requestWithCapacityRetry(() => service.request('POST', body, token, {headers: headers || {}, query, handlesThrottle: true}), { + return requestWithCapacityRetry(() => service.request('POST', body, token, {headers: headers || {}, query, handlesThrottle: true, timeout: HEAVY_REQUEST_TIMEOUT_MS}), { gate: getCapacityGate(service.URL), isCancelled: context.isCancelled || (() => false), maxWaitMs: context.isBulk ? CAPACITY_WAIT_CAP_MS : INTERACTIVE_WAIT_CAP_MS, @@ -2274,6 +2295,7 @@ const MapProject = () => { ...extraParams }, handlesThrottle: true, + timeout: HEAVY_REQUEST_TIMEOUT_MS, } ), retryOptions: {gate: getCapacityGate(service.URL), onCapacity: noteCapacity}, @@ -2596,6 +2618,8 @@ const MapProject = () => { } const fetchBulkBridgeCandidates = async (_rows, algo) => { + // A row answered after a newer run started belongs to no run. + const runTicket = autoMatchRunTicketRef.current setLoadingMatches(true) setBridgeCandidatesStartedAt(moment()) for (let index = 0; index < _rows.length; index++) { @@ -2612,6 +2636,8 @@ const MapProject = () => { // Stop doesn't wait on a request that never answers (ocl_issues#2849). await untilCancelled(fetchBridgeCandidates(_rows[index], 0, undefined, undefined, undefined, false, true, ((response, payload) => { + if(runTicket !== autoMatchRunTicketRef.current) + return const index = payload.rows[0].__index const results = (isArray(response) ? response : response?.data) log({action: 'algo_finished', extras: getAlgoLogExtras(algo)}, index) @@ -2637,6 +2663,8 @@ const MapProject = () => { } const fetchBulkScispacyCandidates = async (_rows, algo) => { + // A row answered after a newer run started belongs to no run. + const runTicket = autoMatchRunTicketRef.current setLoadingMatches(true) setScispacyCandidatesStartedAt(moment()) for (let index = 0; index < _rows.length; index++) { @@ -2654,6 +2682,8 @@ const MapProject = () => { setLoadingMatches(true) // Stop doesn't wait on a request that never answers (ocl_issues#2849). await untilCancelled(fetchScispacyCandidates(_rows[index], false, false, true, (response => { + if(runTicket !== autoMatchRunTicketRef.current) + return const _index = _rows[index].__index const results = [{row: _rows[index], results: fromScispacyResultsToConcepts(getScispacyRowResults(response.data, _index))}] log({action: 'algo_finished', extras: getAlgoLogExtras(algo)}, _index) @@ -3741,6 +3771,7 @@ const MapProject = () => { encoder_model: !isMultiAlgo && encoderModel ? encoderModel : undefined }, handlesThrottle: true, + timeout: HEAVY_REQUEST_TIMEOUT_MS, } ), {retryOptions: { gate: getCapacityGate(service.URL), @@ -3815,6 +3846,18 @@ const MapProject = () => { } markAlgo(__row.__index, algoId, 0) setIsLoadingInDecisionView(true) + // One algorithm failing doesn't stop the row's others, and the panel + // stops loading (ocl_online#283). ScispaCy and the bridge show their own + // failures, then go on here too (ocl_issues#2849). + const goOnAfterFailure = () => { + setIsLoadingInDecisionView(false) + const nextAlgoAfterFailure = getNextAlgoDef(algoId) + if(nextAlgoAfterFailure?.id && (offset === 0 || nextAlgoAfterFailure.type !== 'ocl-scispacy')) { + markAlgo(__row.__index, nextAlgoAfterFailure.id, 0) + fetchAllCandidatesForRow(nextAlgoAfterFailure.id, __row, offset, _retired, scrollToBottom, _filters, forceReload, true) + } else + scheduleRerank(__row.__index) + } const onResponse = (response, payload) => { const projectContext = buildProjectContext() const logExtras = getAlgoLogExtras(getAlgoDef(algoId)) @@ -3847,15 +3890,7 @@ const MapProject = () => { rawResponse: response })) } - // One algorithm failing doesn't stop the row's others, and the - // panel stops loading (ocl_online#283). - setIsLoadingInDecisionView(false) - const nextAlgoAfterFailure = getNextAlgoDef(algoId) - if(nextAlgoAfterFailure?.id && (offset === 0 || nextAlgoAfterFailure.type !== 'ocl-scispacy')) { - markAlgo(__row.__index, nextAlgoAfterFailure.id, 0) - fetchAllCandidatesForRow(nextAlgoAfterFailure.id, __row, offset, _retired, scrollToBottom, _filters, forceReload, true) - } else - scheduleRerank(__row.__index) + goOnAfterFailure() return } log({action: 'algo_finished', extras: logExtras}, __row.__index) @@ -3929,9 +3964,9 @@ const MapProject = () => { if(['ocl-semantic', 'ocl-search', 'custom'].includes(algoDef.type)) { fetchOCLOrCustomCandidates(algoDef, _row, offset, _retired, _filters, onResponse) } else if (algoDef.type === 'ocl-scispacy') { - fetchScispacyCandidates(__row, scrollToBottom, forceReload, false, onResponse) + fetchScispacyCandidates(__row, scrollToBottom, forceReload, false, onResponse, goOnAfterFailure) } else if (['ocl-bridge', 'ocl-ciel-bridge'].includes(algoDef.type)) { - fetchBridgeCandidates(__row, offset, _retired, scrollToBottom, _filters, forceReload, false, onResponse) + fetchBridgeCandidates(__row, offset, _retired, scrollToBottom, _filters, forceReload, false, onResponse, goOnAfterFailure) } } else { setAlert({message: t('map_project.no_valid_columns_for_matching')}) @@ -3939,7 +3974,9 @@ const MapProject = () => { } } - const fetchScispacyCandidates = async (_row, scrollToBottom, forceReload=false, isBulk=false, callback) => { + // onFailure: a row matched by hand goes on to its next algorithm after + // ScispaCy fails or stays throttled (ocl_issues#2849). + const fetchScispacyCandidates = async (_row, scrollToBottom, forceReload=false, isBulk=false, callback, onFailure) => { let __row = isEmpty(_row) ? row : _row // Gate on AlgorithmResponse presence, not candidate count: a successful // invocation that returned zero matches still writes an algorithm_response @@ -3989,9 +4026,13 @@ const MapProject = () => { // A run's waits end when it stops; a row matched by hand isn't stopped. // (abortRef stays set after a Stop, so it can't gate a manual call.) const isCancelled = isBulk ? getRunStopCheck() : () => false + // A run's call that ends after a newer run started leaves its stages alone. + const runTicket = autoMatchRunTicketRef.current + const isSuperseded = () => isBulk && runTicket !== autoMatchRunTicketRef.current const capacityWait = trackCapacityWait([__row.__index], {algo: SCISPACY_ALGO_ID}) const endStopped = () => { - markAlgo(__row.__index, SCISPACY_ALGO_ID, -1) + if(!isSuperseded()) + markAlgo(__row.__index, SCISPACY_ALGO_ID, -1) setIsLoadingInDecisionView(false) } @@ -4007,6 +4048,7 @@ const MapProject = () => { // scalar row_index; isBulk distinguishes a run call from a manual one. headers: attrHeaders({rowIndex: __row.__index, algorithmId: SCISPACY_ALGO_ID, isRunTraffic: isBulk}), handlesThrottle: true, + timeout: HEAVY_REQUEST_TIMEOUT_MS, }), { gate: getCapacityGate(service.URL), isCancelled, @@ -4030,6 +4072,8 @@ const MapProject = () => { return } if(result.reason === 'throttled') { + if(isSuperseded()) + return // Not run, not failed: the row can be run again. markAlgo(__row.__index, SCISPACY_ALGO_ID, -4) log({action: 'algo_throttled', description: t('map_project.row_throttled'), extras: {algo: SCISPACY_ALGO_ID, attempts: result.attempts, waited_ms: result.waitedMs}}, __row.__index) @@ -4037,10 +4081,13 @@ const MapProject = () => { noteThrottledRunRow(SCISPACY_ALGO_ID, __row.__index) setAlert(prev => (prev?.severity === 'error' ? prev : {message: t('map_project.algorithm_throttled'), severity: 'warning'})) setIsLoadingInDecisionView(false) + onFailure?.() return } const err = result.error + if(isSuperseded()) + return const body = err?.response?.data // The service's host is booting: it answers 503 warming_up, then 502 // until the app is up. @@ -4052,6 +4099,7 @@ const MapProject = () => { log({action: 'algo_failed', extras: {algo: SCISPACY_ALGO_ID, status: 503, detail: 'warming_up_timeout'}}, __row.__index) setAlert({message: "OCL's scispacy service did not come up within 10 minutes.", severity: 'error'}) setIsLoadingInDecisionView(false) + onFailure?.() return body } // Don't cover an error an earlier algorithm for this row showed. @@ -4071,6 +4119,7 @@ const MapProject = () => { log({action: 'algo_failed', extras: {algo: SCISPACY_ALGO_ID, status: err?.response?.status || null, detail: body?.detail, error: err?.message}}, __row.__index) setAlert({message: body?.detail || SCISPACY_ERROR_MESSAGE, severity: 'warning'}) setIsLoadingInDecisionView(false) + onFailure?.() return body } } @@ -4149,17 +4198,29 @@ const MapProject = () => { const rerank = async (_index, isBulk=false, isRunTraffic=isBulk) => { const index = isNumber(_index) ? _index : rowIndex if(!isNumber(index)) return null + // A run's rerank ends when the run stops, and a stopped run proposes no + // more mappings; a row reranked by hand isn't stopped (ocl_issues#2849). + const isStopped = isRunTraffic ? getRunStopCheck() : () => false + // A proposal lands a second later: skip it if the run stopped meanwhile. + const proposeMapping = () => setTimeout(() => { + if(!isStopped()) + setAutoMatched([index]) + }, 1000) const rerankInFlight = inFlightRerankRef.current.get(index) if(rerankInFlight) { - // Another rerank is in flight, or waiting its turn, for this row; flag - // a rerun. The run's sweep waits for it, then proposes the row's - // mapping, which only the sweep's own call does (ocl_issues#2849). - rerankRerunNeededRef.current.add(index) - if(!isBulk) + if(!isBulk) { + // Another rerank is in flight, or waiting its turn, for this row: + // flag a rerun for the candidates that arrive meanwhile. + rerankRerunNeededRef.current.add(index) return null + } + // The run's sweep waits for it, then runs again for any candidates + // that arrived meanwhile, and proposes the row's mapping, which only + // the sweep's own call does (ocl_issues#2849). await rerankInFlight - setTimeout(() => setAutoMatched([index]), 1000) - return null + if(isStopped()) + return null + return rerank(index, true, isRunTraffic) } // Wait for any in-flight $lookups for this row's concepts. buildRerankRowsForRow // filters out nameless entries, so running before lookups complete silently @@ -4192,7 +4253,7 @@ const MapProject = () => { // Without the setAutoMatched trigger, auto-match would never propose // a mapping even for rows with a clearly-recommended top candidate. if(isBulk && isNumber(index)) - setTimeout(() => setAutoMatched([index]), 1000) + proposeMapping() return null } // A rerank in this run already stayed refused for the whole cap: don't @@ -4205,8 +4266,7 @@ const MapProject = () => { inFlightRerankRef.current.set(index, new Promise(resolve => { settleInFlight = resolve })) markAlgo(index, 'rerank', 0) const service = APIService.concepts().appendToUrl('$rerank/') - // A run's rerank ends when the run stops; a row reranked by hand doesn't. - const isCancelled = isRunTraffic ? getRunStopCheck() : () => false + const isCancelled = isStopped // A run's rerank that ends after a newer run started leaves its stages alone. const runTicket = autoMatchRunTicketRef.current const isSuperseded = () => isRunTraffic && runTicket !== autoMatchRunTicketRef.current @@ -4230,7 +4290,7 @@ const MapProject = () => { q: query, rows: latestRerankRows.length ? latestRerankRows : rerankRows, ...(encoderModel ? { encoder_model: encoderModel } : {}) - }, null, {headers: attrHeaders({rowIndex: index, algorithmId: 'reranker', isRunTraffic}), handlesThrottle: true}), { + }, null, {headers: attrHeaders({rowIndex: index, algorithmId: 'reranker', isRunTraffic}), handlesThrottle: true, timeout: HEAVY_REQUEST_TIMEOUT_MS}), { gate: getCapacityGate(service.URL), isCancelled, onCapacity: noteCapacity, @@ -4252,6 +4312,8 @@ const MapProject = () => { return null } if(!result.ok) { + if(isSuperseded()) + return null if(handlePreviewLimitError(result.error, index, 'rerank', {isRun: isRunTraffic, stopPhase: 'rerank'})) return null log({action: 'rerank_failed', description: `Rerank failed with ${encoderModel}`, extras: {status: result.error?.response?.status || null, error: result.error?.response?.data?.detail || result.error?.message || null}}, index) @@ -4296,10 +4358,12 @@ const MapProject = () => { // (Removed legacy allCandidates re-stamp: rerank_score now lives on // rowMatchState[idx].concept_rows[concept_key], which we updated above.) - markAlgo(index, 'rerank', 1) + // The scores stand either way; a newer run's stages are its own. + if(!isSuperseded()) + markAlgo(index, 'rerank', 1) log({action: 'rerank_finished', description: `Reranked with ${encoderModel}`}, index) if(isBulk) - setTimeout(() => setAutoMatched([index]), 1000) + proposeMapping() return response } catch (e) { log({action: 'rerank_failed', description: `Rerank failed with ${encoderModel}`, extras: {error: e?.message || null}}, index) @@ -4511,7 +4575,9 @@ const MapProject = () => { return formatted } - const fetchBridgeCandidates = (_row, offset=0, _retired, scrollToBottom, _filters, forceReload=false, isBulk=false, callback) => { + // onFailure: a row matched by hand goes on to its next algorithm after the + // bridge fails or stays throttled (ocl_issues#2849). + const fetchBridgeCandidates = (_row, offset=0, _retired, scrollToBottom, _filters, forceReload=false, isBulk=false, callback, onFailure) => { let __row = isEmpty(_row) ? row : _row const bridgeAlgoId = bridgeAlgo?.id || 'ocl-ciel-bridge' const rowEntry = rowMatchStateRef.current?.[__row.__index] @@ -4542,6 +4608,9 @@ const MapProject = () => { // context tells that post() the row, and whether a Stop ends its waits // (ocl_issues#2849). const context = {rowIndex: __row.__index, isBulk, isCancelled: isBulk ? getRunStopCheck() : () => false, posted: false} + // A run's call that ends after a newer run started leaves its stages alone. + const runTicket = autoMatchRunTicketRef.current + const isSuperseded = () => isBulk && runTicket !== autoMatchRunTicketRef.current bridgeCallContextRef.current = context try { bridgeRef.current?.fetchBridgeCandidates( @@ -4565,6 +4634,10 @@ const MapProject = () => { resolve() }, (response, errorMsg) => { + if(isSuperseded()) { + resolve() + return + } if(response?.cancelled) { markAlgo(__row.__index, bridgeAlgoId, -1) setIsLoadingInDecisionView(false) @@ -4580,6 +4653,7 @@ const MapProject = () => { noteThrottledRunRow(bridgeAlgoId, __row.__index) setAlert(prev => (prev?.severity === 'error' ? prev : {message: response.detail, severity: 'warning'})) setIsLoadingInDecisionView(false) + onFailure?.() resolve() return } @@ -4593,6 +4667,7 @@ const MapProject = () => { log({action: 'algo_failed', extras: getAlgoLogExtras(bridgeAlgo)}, __row.__index) setAlert({message: response?.detail || errorMsg, severity: 'error'}) setIsLoadingInDecisionView(false) + onFailure?.() resolve() }, // ocl_online#105 Phase 5: attribution headers for the bridge $match, @@ -4603,6 +4678,7 @@ const MapProject = () => { ) } catch (err) { failBridgeRow(err?.message || t('unknown_error')) + onFailure?.() resolve() return } finally { @@ -4615,6 +4691,7 @@ const MapProject = () => { markAlgo(__row.__index, bridgeAlgoId, -3) log({action: 'algo_skipped', extras: {...getAlgoLogExtras(bridgeAlgo), reason: 'bridge_unavailable'}}, __row.__index) setIsLoadingInDecisionView(false) + onFailure?.() resolve() } }) @@ -4765,7 +4842,7 @@ const MapProject = () => { // A busy server's 429 is waited out, and a gateway error retried // (ocl_issues#2849); a 404 is retried below. const result = await requestWithCapacityRetry( - () => service.request('GET', undefined, authToken, {query: {includeMappings: true, mappingBrief: true, mapTypes: 'SAME-AS,SAME AS,SAME_AS', verbose: true}, headers: attrHeaders({rowIndex: logRowIndex, isRunTraffic}), handlesThrottle: true}), + () => service.request('GET', undefined, authToken, {query: {includeMappings: true, mappingBrief: true, mapTypes: 'SAME-AS,SAME AS,SAME_AS', verbose: true}, headers: attrHeaders({rowIndex: logRowIndex, isRunTraffic}), handlesThrottle: true, timeout: LIGHT_REQUEST_TIMEOUT_MS}), {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, onCapacity: noteCapacity} ) if(!result.ok) @@ -4834,6 +4911,7 @@ const MapProject = () => { headers: attrHeaders({rowIndex: logRowIndex, isRunTraffic}), query: resolveNamespace ? {namespace: resolveNamespace} : {}, handlesThrottle: true, + timeout: LIGHT_REQUEST_TIMEOUT_MS, }), {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, onCapacity: noteCapacity} ) @@ -4941,7 +5019,7 @@ const MapProject = () => { // which then threw here. A 429 is waited out; an error says so // (ocl_issues#2849). requestWithCapacityRetry( - () => getLookupService().request('GET', undefined, lookupConfig?.token, {query, handlesThrottle: true}), + () => getLookupService().request('GET', undefined, lookupConfig?.token, {query, handlesThrottle: true, timeout: LIGHT_REQUEST_TIMEOUT_MS}), {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, onCapacity: noteCapacity, ...trackCapacityWait([])} ).then(result => { if(!result.ok) { @@ -5002,7 +5080,7 @@ const MapProject = () => { // A failed facets call isn't cached, so the next look tries again // (ocl_issues#2849). const request = requestWithCapacityRetry( - () => getLookupService().request('GET', undefined, lookupConfig?.token, {query, handlesThrottle: true}), + () => getLookupService().request('GET', undefined, lookupConfig?.token, {query, handlesThrottle: true, timeout: LIGHT_REQUEST_TIMEOUT_MS}), {maxWaitMs: INTERACTIVE_WAIT_CAP_MS, onCapacity: noteCapacity, ...trackCapacityWait([])} ).then(result => { if(!result.ok) @@ -5872,15 +5950,18 @@ const MapProject = () => { { // A request is waiting out a busy server: the run is // slower, not stuck (ocl_issues#2849). + // Short in the split view, where the full notice would be cut off. capacityWaitRows && - } - color='warning' - variant='outlined' - size='small' - label={t('map_project.waiting_for_capacity')} - sx={{margin: '5px'}} - /> + + } + color='warning' + variant='outlined' + size='small' + label={isSplitView ? t('map_project.waiting_for_capacity_short') : t('map_project.waiting_for_capacity')} + sx={{margin: '5px'}} + /> + } } diff --git a/src/components/map-projects/__tests__/matchBatch.test.js b/src/components/map-projects/__tests__/matchBatch.test.js index 14ad45f1..4d4a24b7 100644 --- a/src/components/map-projects/__tests__/matchBatch.test.js +++ b/src/components/map-projects/__tests__/matchBatch.test.js @@ -538,3 +538,17 @@ test('runWithConcurrency: a pause longer than maxHoldMs doesn\'t hold the queue assert.deepEqual(ran, [1, 2, 3]) assert.ok(Date.now() - started < 1000) }) + +// Codex review, pass 2: the scheduler's hold budget is cumulative. +test('runWithConcurrency: a pause that keeps extending holds the queue for maxHoldMs in all, then lets it go', async () => { + const gate = createCapacityGate() + gate.pause(40) + // other requests keep pushing the pause out + const timer = setInterval(() => gate.pause(40), 10) + const started = Date.now() + const ran = [] + await runWithConcurrency([1, 2], {concurrency: 1, gate, maxHoldMs: 150, pollMs: 5, run: async () => { ran.push(Date.now() - started) }}) + clearInterval(timer) + assert.equal(ran.length, 2) + assert.ok(ran[0] >= 140 && ran[0] < 1000, `first batch went out at ${ran[0]} ms`) +}) diff --git a/src/components/map-projects/matchBatch.js b/src/components/map-projects/matchBatch.js index 69d274d7..1352f682 100644 --- a/src/components/map-projects/matchBatch.js +++ b/src/components/map-projects/matchBatch.js @@ -103,9 +103,10 @@ export const runMatchBatch = async ({ * The bulk Auto Match scheduler: runs each item through run(item) with at most * concurrency in flight. While the gate is paused (a request got a 429) it * sends nothing new, and waits for the gate to reopen or a request in flight - * to settle, so a throttled run doesn't drain its queue into more refusals. A - * pause longer than maxHoldMs doesn't hold the queue: its items go out and - * end throttled at once, without sending. + * to settle, so a throttled run doesn't drain its queue into more refusals. + * It holds the queue for at most maxHoldMs in all, however often the pause is + * extended; after that, and for a pause longer than what's left of it, the + * items go out and wait (or end throttled) as requests of their own. * * shouldAbort stops it at once, even while a request in flight never answers * (those are left to finish on their own), and returns false. shouldSkipRest @@ -117,7 +118,8 @@ export const runWithConcurrency = async (items, { }) => { const queue = items.slice() const active = new Set() - const isHeld = () => Boolean(gate?.isPaused() && gate.pausedForMs() <= maxHoldMs) + let heldMs = 0 + const isHeld = () => Boolean(gate?.isPaused() && gate.pausedForMs() <= maxHoldMs - heldMs) while(queue.length || active.size) { while(queue.length && active.size < concurrency) { if(shouldAbort()) @@ -136,11 +138,15 @@ export const runWithConcurrency = async (items, { active.add(promise) } const waitingOn = [...active] - if(queue.length && active.size < concurrency && isHeld()) - waitingOn.push(gate.wait(shouldAbort, {pollMs, maxMs: maxHoldMs})) + const holding = queue.length && active.size < concurrency && isHeld() + if(holding) + waitingOn.push(gate.wait(shouldAbort, {pollMs, maxMs: maxHoldMs - heldMs})) if(!waitingOn.length) continue + const waitStartedAt = Date.now() const { settled } = await untilCancelled(Promise.race(waitingOn), shouldAbort, {pollMs}) + if(holding) + heldMs += Date.now() - waitStartedAt if(!settled) return false } diff --git a/src/i18n/locales/en/translations.json b/src/i18n/locales/en/translations.json index 388a011b..7a089dbf 100644 --- a/src/i18n/locales/en/translations.json +++ b/src/i18n/locales/en/translations.json @@ -590,6 +590,7 @@ "auto_match_rows_failed_algorithm": "{{algorithm}} on row(s) {{rows}}", "match_request_failed": "Couldn't get candidates ({{error}}). Try again.", "waiting_for_capacity": "Waiting for capacity: results may take longer due to demand", + "waiting_for_capacity_short": "Waiting for capacity", "row_throttled": "Server busy: not matched yet. Retry later", "algorithm_throttled": "The server is busy, so this row wasn't matched this time. Nothing failed: run it again later.", "auto_match_rows_throttled": "The server was busy, so {{count}} input row(s) weren't matched: {{details}}. Nothing failed. Run Auto Match on those rows again later to finish them.", diff --git a/src/i18n/locales/es/translations.json b/src/i18n/locales/es/translations.json index 9e549a5f..bc856dbd 100644 --- a/src/i18n/locales/es/translations.json +++ b/src/i18n/locales/es/translations.json @@ -554,6 +554,7 @@ "auto_match_rows_failed_algorithm": "{{algorithm}} en la(s) fila(s) {{rows}}", "match_request_failed": "No se pudieron obtener candidatos ({{error}}). Inténtalo de nuevo.", "waiting_for_capacity": "Esperando capacidad: los resultados pueden tardar más debido a la demanda", + "waiting_for_capacity_short": "Esperando capacidad", "row_throttled": "Servidor ocupado: aún sin procesar. Reintenta más tarde", "algorithm_throttled": "El servidor está ocupado, así que esta fila no se procesó esta vez. Nada falló: vuelve a ejecutarla más tarde.", "auto_match_rows_throttled": "El servidor estaba ocupado, así que {{count}} fila(s) de entrada no se procesaron: {{details}}. Nada falló. Vuelve a ejecutar Auto Match en esas filas más tarde para terminarlas.", diff --git a/src/i18n/locales/zh/translations.json b/src/i18n/locales/zh/translations.json index 3eb220ce..a835a20e 100644 --- a/src/i18n/locales/zh/translations.json +++ b/src/i18n/locales/zh/translations.json @@ -579,6 +579,7 @@ "auto_match_rows_failed_algorithm": "{{algorithm}}:第 {{rows}} 行", "match_request_failed": "无法获取候选项({{error}})。请重试。", "waiting_for_capacity": "正在等待容量:由于需求量大,结果可能需要更长时间", + "waiting_for_capacity_short": "正在等待容量", "row_throttled": "服务器繁忙:尚未匹配。请稍后重试", "algorithm_throttled": "服务器繁忙,本次未能匹配此行。没有出错:请稍后重新运行。", "auto_match_rows_throttled": "服务器繁忙,{{count}} 个输入行未能匹配:{{details}}。没有出错。请稍后对这些行重新运行自动匹配以完成匹配。", diff --git a/src/services/__tests__/capacity.test.js b/src/services/__tests__/capacity.test.js index 841d7da1..062d8e5d 100644 --- a/src/services/__tests__/capacity.test.js +++ b/src/services/__tests__/capacity.test.js @@ -21,6 +21,8 @@ import { createLimiter, getCapacityHeaders, getRetryAfterMs, + HEAVY_REQUEST_TIMEOUT_MS, + LIGHT_REQUEST_TIMEOUT_MS, requestWithCapacityRetry, sleepUnlessCancelled, untilCancelled, @@ -511,19 +513,22 @@ test('requestWithCapacityRetry: a gate pause that keeps extending ends "throttle assert.ok(clock.t <= 60250, `waited ${clock.t}`) }) -test('requestWithCapacityRetry: a Retry-After past the cap still pauses the gate, so the other requests hold back too', async () => { +// The e2e re-run (scenario T): pausing the whole server for a Retry-After past +// the cap blocked every other algorithm, rerank and later run in the tab for +// as long. The request gives up alone; the run stops asking for that +// algorithm (MapProject's run-level throttle), and the others carry on. +test('requestWithCapacityRetry: a Retry-After past the cap ends "throttled" without pausing the gate for the others', async () => { const clock = virtualClock() const gate = createCapacityGate({now: clock.now}) - const first = await requestWithCapacityRetry(scripted(throttled(86400)).send, {gate, now: clock.now, sleep: clock.sleep}) - const second = scripted(ok()) - const next = await requestWithCapacityRetry(second.send, {gate, now: clock.now, sleep: clock.sleep}) + const first = await requestWithCapacityRetry(scripted(throttled(4000)).send, {gate, now: clock.now, sleep: clock.sleep}) + const other = scripted(ok()) + const next = await requestWithCapacityRetry(other.send, {gate, now: clock.now, sleep: clock.sleep}) assert.equal(first.reason, 'throttled') - assert.equal(gate.pausedForMs(), 86400 * 1000) - // the next request doesn't ask again: the server said when to come back - assert.equal(next.reason, 'throttled') - assert.equal(second.sent.length, 0) + assert.equal(gate.isPaused(), false) + assert.equal(next.ok, true) + assert.equal(other.sent.length, 1) assert.equal(clock.t, 0) }) @@ -539,3 +544,17 @@ test('untilCancelled: the promise\'s value when it settles, or cancelled when St test('untilCancelled: a rejection passes through', async () => { await assert.rejects(untilCancelled(Promise.reject(new Error('boom')), () => false, {pollMs: 5}), /boom/) }) + +// Codex review, pass 2: transport timeouts bound a send that never answers. +test('request timeouts: heavy calls outlast any answer the API gives; light calls give up within a minute', () => { + assert.ok(HEAVY_REQUEST_TIMEOUT_MS > 10 * 60 * 1000) + assert.equal(LIGHT_REQUEST_TIMEOUT_MS, 60 * 1000) +}) + +test('requestWithCapacityRetry: an attempt that timed out (axios ECONNABORTED) is retried like a network error', async () => { + const timedOut = Object.assign(new Error('timeout of 60000ms exceeded'), {isAxiosError: true, code: 'ECONNABORTED'}) + const {send, sent} = scripted(timedOut, ok()) + const result = await requestWithCapacityRetry(send, {sleep: async () => {}}) + assert.equal(result.ok, true) + assert.equal(sent.length, 2) +}) diff --git a/src/services/capacity.js b/src/services/capacity.js index 0953cc87..b4e0ab42 100644 --- a/src/services/capacity.js +++ b/src/services/capacity.js @@ -24,6 +24,14 @@ const MAX_ERROR_RETRY_AFTER_MS = 60000 // to listen to. const POLL_MS = 250 +// Transport timeouts: a send that never answers can't hold a run, a rerank +// slot or the logs forever. A heavy call ($match, $rerank) gets longer than +// the API itself runs a request before giving up on it, so a timeout never +// cuts short (and re-sends, and re-meters) work the server is still doing. +// Light calls (logs, run records, lookups, search) answer within seconds. +export const HEAVY_REQUEST_TIMEOUT_MS = 11 * 60 * 1000 +export const LIGHT_REQUEST_TIMEOUT_MS = 60 * 1000 + // The gateway answers these while the API rolls out or a task restarts. A 500 // comes from the API itself, and retrying it only adds load. const RETRYABLE_STATUSES = new Set([502, 503, 504]) @@ -238,11 +246,13 @@ export const requestWithCapacityRetry = async (send, { Math.max(retryAfterMs, MIN_THROTTLE_WAIT_MS) throttles += 1 const delayMs = baseMs * (1 + random() * THROTTLE_JITTER) - // The other requests to this server hold back as long, even when - // this one gives up now. - gate?.pause(baseMs) + // Past the cap, this request gives up alone. Pausing the gate for so + // long would hold back every other request to the server (another + // algorithm, rerank, the row panel) for as long; the caller stops + // asking for what was refused instead. if(waitedMs + delayMs > maxWaitMs) return end({ ok: false, reason: 'throttled', error: err }) + gate?.pause(baseMs) if(!(await waitFor(delayMs, { reason: 'throttled', status: 429, retryAfterMs, capacity: getCapacityHeaders(err.response) }))) return cancelled() continue From 0d0221d745d01dd1c830b815e3877f1f10e832f3 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Tue, 29 Sep 2026 23:35:25 -0400 Subject: [PATCH 05/14] closes OpenConceptLab/ocl_issues#2849 | Codex pass 3: a superseded run's rerank scores aren't merged, and the orphan-run close retries - A rerank from a run that a newer run replaced doesn't write its scores: the newer run may rank with another encoder or input, and would otherwise skip those candidates as already scored. - Closing a run the server opened after the user stopped retries a busy server or a gateway error, like the normal close, instead of sending once. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index 1910f1b4..e9b63a3a 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -1875,12 +1875,15 @@ const MapProject = () => { const outcome = await untilCancelled(request, isStopped) if(!outcome.settled) { // Stopped before the server answered: close a run it opens anyway. + // Idempotent, so retried like the normal close; never rejects. request.then(late => { const lateId = late.ok && late.response?.data?.id if(lateId) - APIService.new().overrideURL('/auto-match-runs/' + lateId + '/') - .request('PATCH', {completed_rows: 0, failed_rows: 0, completion_status: 'cancelled'}, null, {handlesThrottle: true, timeout: LIGHT_REQUEST_TIMEOUT_MS}) - .catch(() => {}) + requestWithCapacityRetry( + () => APIService.new().overrideURL('/auto-match-runs/' + lateId + '/') + .request('PATCH', {completed_rows: 0, failed_rows: 0, completion_status: 'cancelled'}, null, {handlesThrottle: true, timeout: LIGHT_REQUEST_TIMEOUT_MS}), + {maxWaitMs: INTERACTIVE_WAIT_CAP_MS} + ) }) return } @@ -4321,6 +4324,10 @@ const MapProject = () => { return null } const response = result.response + // A newer run may rank with another encoder or input: its scores are + // its own (Codex pass 3). + if(isSuperseded()) + return null // Write rerank_score into the row's ConceptRows. matchRerankResultToKey // throws on canonical-identity miss; surface to the alert state so a From f22d7d7924ce9181bdbf8b87024cbb4bf3c5d779 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Tue, 29 Sep 2026 23:40:46 -0400 Subject: [PATCH 06/14] OpenConceptLab/ocl_issues#2849 | Codex pass 4: a stopped rerank can't touch the next run, and a run with throttled reranks is partial - rerank() captures its run when called. After waiting for a row's lookups it checks Stop before touching anything, joins a rerank that started for the row meanwhile, and only clears its own in-flight entry. A stopped run's rerank could otherwise mark a newer run's stage. - A run whose matching finished but whose reranks stayed throttled is closed "partial", not "completed"; the row counts stay as they were. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 22 ++++++++++++++++------ src/services/__tests__/attribution.test.js | 8 ++++++++ src/services/attribution.js | 3 ++- 3 files changed, 26 insertions(+), 7 deletions(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index e9b63a3a..12750439 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -4204,6 +4204,10 @@ const MapProject = () => { // A run's rerank ends when the run stops, and a stopped run proposes no // more mappings; a row reranked by hand isn't stopped (ocl_issues#2849). const isStopped = isRunTraffic ? getRunStopCheck() : () => false + // The run this call belongs to, captured now: a newer run's stages are + // its own (ocl_issues#2849). + const runTicket = autoMatchRunTicketRef.current + const isSuperseded = () => isRunTraffic && runTicket !== autoMatchRunTicketRef.current // A proposal lands a second later: skip it if the run stopped meanwhile. const proposeMapping = () => setTimeout(() => { if(!isStopped()) @@ -4236,6 +4240,13 @@ const MapProject = () => { // Bounded, so a lookup that never settles can't stall rerank, and // with it a run (ocl_online#283). await waitForLookups(pendingLookups, RERANK_LOOKUP_WAIT_MS, stuckLookupsRef.current) + // The run may have stopped, and another started, during the wait: + // touch nothing of the new run's (ocl_issues#2849). + if(isStopped()) + return null + // Another rerank for the row may have started meanwhile: join it. + if(inFlightRerankRef.current.has(index)) + return rerank(index, isBulk, isRunTraffic) // The settling lookups fired scheduleRerank via settle()/writeConceptCachePatch, // setting a 300ms debounce timer. Clear it — we're already inside rerank() and // about to run, so that timer would only queue a redundant rerankRerunNeeded. @@ -4266,13 +4277,11 @@ const MapProject = () => { return null } let settleInFlight - inFlightRerankRef.current.set(index, new Promise(resolve => { settleInFlight = resolve })) + const ownInFlight = new Promise(resolve => { settleInFlight = resolve }) + inFlightRerankRef.current.set(index, ownInFlight) markAlgo(index, 'rerank', 0) const service = APIService.concepts().appendToUrl('$rerank/') const isCancelled = isStopped - // A run's rerank that ends after a newer run started leaves its stages alone. - const runTicket = autoMatchRunTicketRef.current - const isSuperseded = () => isRunTraffic && runTicket !== autoMatchRunTicketRef.current let release = null try { // At most RERANK_MAX_IN_FLIGHT $rerank calls at once, whatever fired @@ -4325,7 +4334,7 @@ const MapProject = () => { } const response = result.response // A newer run may rank with another encoder or input: its scores are - // its own (Codex pass 3). + // its own (ocl_issues#2849). if(isSuperseded()) return null @@ -4378,7 +4387,8 @@ const MapProject = () => { return null } finally { release?.() - inFlightRerankRef.current.delete(index) + if(inFlightRerankRef.current.get(index) === ownInFlight) + inFlightRerankRef.current.delete(index) settleInFlight() // If new ConceptRows arrived while we were in flight, fire again. if(rerankRerunNeededRef.current.has(index)) { diff --git a/src/services/__tests__/attribution.test.js b/src/services/__tests__/attribution.test.js index a9362e80..e2f12b86 100644 --- a/src/services/__tests__/attribution.test.js +++ b/src/services/__tests__/attribution.test.js @@ -231,3 +231,11 @@ test('a user cancel wins over throttled rows', () => { const r = summarizeRunCompletion({ rowStages: {0: {'ocl-search': -4}}, rowIndices: [0], algoIds: ['ocl-search'], aborted: true }) assert.equal(r.completion_status, 'cancelled') }) + +// Codex pass 4: a run whose matching finished but whose reranks stayed +// throttled isn't complete either. +test('a throttled rerank (-4) makes the run partial; the row still counts as completed', () => { + const rowStages = { 0: {'ocl-search': 1, rerank: -4}, 1: {'ocl-search': 1, rerank: 1} } + const r = summarizeRunCompletion({ rowStages, rowIndices: [0, 1], algoIds: ['ocl-search'] }) + assert.deepEqual(r, { completed_rows: 2, failed_rows: 0, completion_status: 'partial' }) +}) diff --git a/src/services/attribution.js b/src/services/attribution.js index 8502cd39..ad225d56 100644 --- a/src/services/attribution.js +++ b/src/services/attribution.js @@ -154,7 +154,8 @@ export const summarizeRunCompletion = ({ const stage = rowStages[idx] || {} const attempted = algoIds.map(id => stage[id]).filter(s => s !== undefined && s !== -1) if(!attempted.length) return - if(attempted.some(s => s === -4)) throttled += 1 + // A row whose rerank stayed throttled is matched but not ranked. + if(attempted.some(s => s === -4) || stage.rerank === -4) throttled += 1 if(attempted.some(s => s === 1)) completed += 1 else if(attempted.every(s => s === -2)) failed += 1 }) From 063f85f963dfe0499bab5d4f7dfeb92ccf83f149 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Tue, 29 Sep 2026 23:46:06 -0400 Subject: [PATCH 07/14] OpenConceptLab/ocl_issues#2849 | Codex pass 5: queued reranks respect the run's throttle cutoff, and a stale throttled rerank is reset - A rerank checks the run's "stopped asking" rule again once it gets its limiter slot: reranks already queued when another one hit the cap end throttled instead of each starting a fresh 30-minute wait. - A run resets a rerank that an earlier run left throttled, and a row whose candidates all have scores already (inline, from a single-algorithm $match) has its rerank marked done; a successful retry no longer shows "Server busy" or closes the run as partial. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index 12750439..c3c3f25e 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -2479,6 +2479,9 @@ const MapProject = () => { const rowId = row.__index const rowState = { ...(next[rowId] || {}) } _selectedAlgos.forEach(algo => { rowState[algo.id] = -1 }) + // A rerank an earlier run left throttled is this run's to retry. + if(rowState.rerank === -4) + rowState.rerank = -1 next[rowId] = rowState }) rowStageRef.current = next @@ -4266,6 +4269,11 @@ const MapProject = () => { // scheduleRerank that already scored the row from algo onResponse. // Without the setAutoMatched trigger, auto-match would never propose // a mapping even for rows with a clearly-recommended top candidate. + // Every candidate already has a score (e.g. inline, from a + // single-algorithm $match): the row's rerank is done. + const conceptRows = Object.values(rowMatchStateRef.current[index]?.concept_rows || {}) + if(query && conceptRows.length && conceptRows.every(cr => isNumber(cr?.rerank_score)) && rowStageRef.current[index]?.rerank !== 1 && !isSuperseded()) + markAlgo(index, 'rerank', 1) if(isBulk && isNumber(index)) proposeMapping() return null @@ -4294,6 +4302,12 @@ const MapProject = () => { markAlgo(index, 'rerank', -1) return null } + // Another of the run's reranks stayed refused for the whole cap while + // this one waited its turn: don't ask again. + if(isRunTraffic && isRunThrottled('rerank')) { + markRowsThrottled([index], 'rerank', {algo: 'reranker'}) + return null + } // Score the candidates that arrived while this call waited its turn too. const latestRerankRows = buildRerankRowsForRow(index) // service.request rejects on a non-2xx; service.post resolved a 5xx as if From 891fa8a8bc7ca112030a2323460f78b811c47d81 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Tue, 29 Sep 2026 23:51:24 -0400 Subject: [PATCH 08/14] OpenConceptLab/ocl_issues#2849 | Codex pass 6: results that land after Stop are saved, and a superseded batch's preview limit can't stop the next run - A batch, bridge, ScispaCy or rerank result that lands after the user stopped its run is still kept (a graceful stop), and now schedules an autosave: the stopped run's own autosave may already have gone. - A preview-limit refusal from a batch of a run that a newer run replaced no longer sets the newer run's quota stop or opens its dialog. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 20 +++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index c3c3f25e..019d9c20 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -1804,6 +1804,14 @@ const MapProject = () => { return () => Boolean(abortRef.current || ticket !== autoMatchRunTicketRef.current) } + // A result that lands after the user stopped its run is kept (a graceful + // stop), so save it too: the stopped run's own autosave may have gone + // already (ocl_issues#2849). + const saveLateRunResult = () => { + if(abortRef.current) + scheduleAutoSave('auto_match_stopped') + } + const noteThrottledRunRow = (algoId, index) => { throttledRunRowsRef.current = { ...throttledRunRowsRef.current, @@ -2322,7 +2330,11 @@ const MapProject = () => { }, // Sets matchQuotaStopRef, which stops the rest of the $match requests // but lets the run finish with what it has. - onPreviewLimit: err => handlePreviewLimitError(err), + // Not once a newer run started: its quota state is its own. + onPreviewLimit: err => { + if(isCurrentRun()) + handlePreviewLimitError(err) + }, // A batch waiting to retry stops on a cancel, once another batch has // hit the match quota, or once a newer run has started. shouldStop: () => Boolean( @@ -2385,6 +2397,8 @@ const MapProject = () => { if(!isMultiAlgo) setStateViews(data, _repo) setMatchedConcepts(prev => [...prev, ...data]); + if(Array.isArray(data) && data.length) + saveLateRunResult() }), }) if(!finished) @@ -2644,6 +2658,7 @@ const MapProject = () => { await untilCancelled(fetchBridgeCandidates(_rows[index], 0, undefined, undefined, undefined, false, true, ((response, payload) => { if(runTicket !== autoMatchRunTicketRef.current) return + saveLateRunResult() const index = payload.rows[0].__index const results = (isArray(response) ? response : response?.data) log({action: 'algo_finished', extras: getAlgoLogExtras(algo)}, index) @@ -2690,6 +2705,7 @@ const MapProject = () => { await untilCancelled(fetchScispacyCandidates(_rows[index], false, false, true, (response => { if(runTicket !== autoMatchRunTicketRef.current) return + saveLateRunResult() const _index = _rows[index].__index const results = [{row: _rows[index], results: fromScispacyResultsToConcepts(getScispacyRowResults(response.data, _index))}] log({action: 'algo_finished', extras: getAlgoLogExtras(algo)}, _index) @@ -4391,6 +4407,8 @@ const MapProject = () => { // The scores stand either way; a newer run's stages are its own. if(!isSuperseded()) markAlgo(index, 'rerank', 1) + if(isRunTraffic) + saveLateRunResult() log({action: 'rerank_finished', description: `Reranked with ${encoderModel}`}, index) if(isBulk) proposeMapping() From 70497eae8669094ff3a5d59244e479f6082160ac Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Wed, 30 Sep 2026 00:00:44 -0400 Subject: [PATCH 09/14] OpenConceptLab/ocl_issues#2849 | Codex pass 8: a stopped run can't carry on, or close the next run's record, once a new run starts Stop re-enables Auto Match as soon as the stopped run's scheduler returns, and a new run resets abortRef. The stopped run's pipeline could then carry on: into its next algorithm, its bridge or ScispaCy loop, or its AI step (spending AI calls on its rows); and its finally closed the new run's AutomatchRun record with the old run's counts. - Each run checks its own Stop (isRunStopped: the user stopped it, or a newer run started) in every phase: the batch scheduler, the algorithm loop, the bridge and ScispaCy loops, the rerank sweep, the AI step and the end-of-run notices, logs and autosave. - Each run closes the AutomatchRun record it opened, passed explicitly, and marks it cancelled when it was stopped or superseded. - Only the current run clears the shared loading state. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 90 +++++++++++++--------- 1 file changed, 55 insertions(+), 35 deletions(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index 019d9c20..75cad71f 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -1902,7 +1902,7 @@ const MapProject = () => { const id = data?.id if(id) { automatchRunRef.current = {id, algoIds: map(selectedAlgos, 'id')} - return + return automatchRunRef.current } if(status === 403 && data?.error_code) { abortRef.current = true @@ -1922,15 +1922,17 @@ const MapProject = () => { // over the rows we set out to process; a row counts as failed only when EVERY // attempted algo failed for it. Never throws — a failed PATCH (or a run that // was never created) must not surface to the user. - const completeAutomatchRun = async (selectedAlgos, intendedRows) => { - const run = automatchRunRef.current - automatchRunRef.current = null + // Each run closes its own record, passed in: by the time a stopped run gets + // here, a newer run's record may be the current one (ocl_issues#2849). + const completeAutomatchRun = async (selectedAlgos, intendedRows, run, aborted) => { + if(automatchRunRef.current === run) + automatchRunRef.current = null if(!run?.id) return const payload = summarizeRunCompletion({ rowStages: rowStageRef.current || [], rowIndices: map(intendedRows, '__index'), algoIds: run.algoIds?.length ? run.algoIds : map(selectedAlgos, 'id'), - aborted: abortRef.current, + aborted, stoppedForQuota: Boolean(matchQuotaStopRef.current), }) // Closing the run is idempotent, so a busy server or a gateway error is @@ -2256,12 +2258,17 @@ const MapProject = () => { const runTicket = autoMatchRunGateRef.current.next() autoMatchRunTicketRef.current = runTicket const isCurrentRun = () => autoMatchRunGateRef.current.isCurrent(runTicket) + // This run's Stop: the user stopped it, or a newer run started (which + // resets abortRef). Every phase checks this, not abortRef alone, so a + // stopped run can't carry on once the next one starts (ocl_issues#2849). + const isRunStopped = () => Boolean(abortRef.current || !isCurrentRun()) throttledRunRowsRef.current = {} // Function to process a single batch const processBatch = async (_repo, rowBatch, algo) => { - if (abortRef.current) { - setLoadingMatches(false) + if (isRunStopped()) { + if(isCurrentRun()) + setLoadingMatches(false) return [] }; if (matchQuotaStopRef.current && spendsMatchQuota(algo, {canBridge})) @@ -2354,7 +2361,7 @@ const MapProject = () => { concurrency: concurrentRequests, gate: getCapacityGate(getMatchAPIService(algo).URL), maxHoldMs: CAPACITY_WAIT_CAP_MS, - shouldAbort: () => abortRef.current, + shouldAbort: isRunStopped, shouldSkipRest: () => isQuotaStopped() || isRunThrottled(algo.id), // A quota stop leaves the rest not run, as before; a server that // stayed too busy leaves them throttled. @@ -2401,7 +2408,7 @@ const MapProject = () => { saveLateRunResult() }), }) - if(!finished) + if(!finished && isCurrentRun()) setLoadingMatches(false) }; @@ -2410,11 +2417,11 @@ const MapProject = () => { const processRerankWithConcurrency = async (_rows, maxConcurrent = 2) => { const finished = await runWithConcurrency(_rows, { concurrency: maxConcurrent, - shouldAbort: () => abortRef.current, + shouldAbort: isRunStopped, shouldSkipRest: () => Boolean(rerankQuotaStopRef.current), run: row => rerank(row.__index, true).catch(() => null), }) - if(!finished) + if(!finished && isCurrentRun()) setLoadingMatches(false) }; // Capped users' algorithms carry the settings they run with, so the @@ -2478,10 +2485,10 @@ const MapProject = () => { // ocl_online#105 Phase 5: open the run record, then guarantee it is // closed out (completed / partial / failed / cancelled) via the finally, // even on cancel or an unexpected throw mid-pipeline. - await createAutomatchRun(_selectedAlgos, rowsToProcess) + const runRecord = await createAutomatchRun(_selectedAlgos, rowsToProcess) // False when the run was refused (a preview limit) or stopped before it // began: it changed nothing to save. - const runStarted = !abortRef.current + const runStarted = !isRunStopped() try { bulkMatchAlgoIdsRef.current = map(_selectedAlgos, 'id') // Reset all algo stages to -1 for every row before starting so that @@ -2504,14 +2511,14 @@ const MapProject = () => { isBulkMatchRunningRef.current = true try { for(const algo of _selectedAlgos) { - if(abortRef.current) break + if(isRunStopped()) break if(matchQuotaStopRef.current && spendsMatchQuota(algo, {canBridge})) continue if(['custom', 'ocl-search', 'ocl-semantic'].includes(algo.type)) await processWithConcurrency(repo, algo, rowsToProcess) else if(['ocl-bridge', 'ocl-ciel-bridge'].includes(algo.type) && canBridge) - await fetchBulkBridgeCandidates(rowsToProcess, algo) + await fetchBulkBridgeCandidates(rowsToProcess, algo, isRunStopped) else if(algo.type === 'ocl-scispacy' && canScispacy) - await fetchBulkScispacyCandidates(rowsToProcess, algo) + await fetchBulkScispacyCandidates(rowsToProcess, algo, isRunStopped) } } finally { isBulkMatchRunningRef.current = false @@ -2524,10 +2531,13 @@ const MapProject = () => { rowsToProcess if(_selectedAlgos.length) await processRerankWithConcurrency(finishedRows, 2) - if(inAIAssistantGroup && autoRunAIAnalysis) { + const runAIStep = Boolean(inAIAssistantGroup && autoRunAIAnalysis && !isRunStopped()) + if(runAIStep) await new Promise(resolve => setTimeout(resolve, 1000)) - await runBulkAIAnalysis(finishedRows) - } else { + if(runAIStep && !isRunStopped()) { + await runBulkAIAnalysis(finishedRows, {isStopped: isRunStopped, isCurrent: isCurrentRun}) + } else if(isCurrentRun()) { + // A newer run's loading state is its own. setIsLoadingInDecisionView(false) setLoadingMatches(false) setEndMatchingAt(moment()) @@ -2549,7 +2559,7 @@ const MapProject = () => { })).join('; ') const endOfRunNotices = [] const failedAlgoIds = keys(failedMatchRows) - if(!abortRef.current && failedAlgoIds.length) { + if(!isRunStopped() && failedAlgoIds.length) { const rowsByAlgo = getRowsByAlgo(failedMatchRows) projectLog({action: 'auto_match_rows_failed', extras: {row_indexes_by_algorithm: rowsByAlgo}}) endOfRunNotices.push(t('map_project.auto_match_rows_failed', { @@ -2559,7 +2569,7 @@ const MapProject = () => { } // Rows the server stayed too busy for weren't run: say so apart from // the failures (ocl_issues#2849). - if(!abortRef.current && keys(throttledRunRowsRef.current).length) { + if(!isRunStopped() && keys(throttledRunRowsRef.current).length) { const rowsByAlgo = getRowsByAlgo(throttledRunRowsRef.current) projectLog({action: 'auto_match_rows_throttled', extras: {row_indexes_by_algorithm: rowsByAlgo}}) endOfRunNotices.push(t('map_project.auto_match_rows_throttled', { @@ -2569,9 +2579,9 @@ const MapProject = () => { } if(endOfRunNotices.length) setAlert({severity: 'warning', message: endOfRunNotices.join(' ')}) - if(!abortRef.current && matchQuotaStopRef.current) + if(!isRunStopped() && matchQuotaStopRef.current) projectLog({action: 'auto_match_stopped_for_quota', extras: {reason: matchQuotaStopRef.current}}) - if(!abortRef.current) + if(!isRunStopped()) projectLog({ action: 'auto_match_finished', extras: { @@ -2590,8 +2600,8 @@ const MapProject = () => { // Save what the run matched, also when it was stopped or failed part // way (ocl_issues#2849). if(runStarted) - scheduleAutoSave(abortRef.current ? 'auto_match_stopped' : 'auto_match') - await completeAutomatchRun(_selectedAlgos, rowsToProcess) + scheduleAutoSave(isRunStopped() ? 'auto_match_stopped' : 'auto_match') + await completeAutomatchRun(_selectedAlgos, rowsToProcess, runRecord, isRunStopped()) refreshMapperQuotaCache() } }, 1000) @@ -2601,7 +2611,9 @@ const MapProject = () => { conceptCacheRef.current = conceptCache; }, [conceptCache]); - const runBulkAIAnalysis = async (_rows) => { + // isStopped: the run's Stop; isCurrent: false once a newer run started, whose + // loading state is its own (ocl_issues#2849). + const runBulkAIAnalysis = async (_rows, {isStopped = () => abortRef.current, isCurrent = () => true} = {}) => { GAService.recordActionEvent('MapProject', 'ai_assistant_run', undefined, { mode: 'bulk', row_count: _rows.length @@ -2622,7 +2634,7 @@ const MapProject = () => { } aiFailuresInARowRef.current = 0 for (let index = 0; index < _rows.length; index++) { - if (abortRef.current) break; + if (isStopped()) break; // AI quota exhausted, or the AI service failing row after row: stop calling // the AI step only. Matching itself already finished before this loop // starts, so nothing else aborts. @@ -2630,6 +2642,8 @@ const MapProject = () => { await fetchRecommendation(_rows[index], resolvedPromptTemplate, true); } + if(!isCurrent()) + return const now = moment() setBulkAIAnalysisEndedAt(now) setEndMatchingAt(now) @@ -2637,14 +2651,17 @@ const MapProject = () => { setIsLoadingInDecisionView(false) } - const fetchBulkBridgeCandidates = async (_rows, algo) => { + // isStopped: the run's Stop, which stays true once a newer run starts. + const fetchBulkBridgeCandidates = async (_rows, algo, isStopped = () => abortRef.current) => { // A row answered after a newer run started belongs to no run. const runTicket = autoMatchRunTicketRef.current setLoadingMatches(true) setBridgeCandidatesStartedAt(moment()) for (let index = 0; index < _rows.length; index++) { - if (abortRef.current) { - setLoadingMatches(false) + if (isStopped()) { + // A newer run's loading state is its own. + if(runTicket === autoMatchRunTicketRef.current) + setLoadingMatches(false) break; }; if (matchQuotaStopRef.current) break; @@ -2676,21 +2693,24 @@ const MapProject = () => { rawResponse: response }), {isRunTraffic: true}) } - })), () => abortRef.current); // wait for completion + })), isStopped); // wait for completion await new Promise(resolve => setTimeout(resolve, 200)); // 1s delay } const now = moment() setBridgeCandidatesEndedAt(now) } - const fetchBulkScispacyCandidates = async (_rows, algo) => { + // isStopped: the run's Stop, which stays true once a newer run starts. + const fetchBulkScispacyCandidates = async (_rows, algo, isStopped = () => abortRef.current) => { // A row answered after a newer run started belongs to no run. const runTicket = autoMatchRunTicketRef.current setLoadingMatches(true) setScispacyCandidatesStartedAt(moment()) for (let index = 0; index < _rows.length; index++) { - if (abortRef.current) { - setLoadingMatches(false) + if (isStopped()) { + // A newer run's loading state is its own. + if(runTicket === autoMatchRunTicketRef.current) + setLoadingMatches(false) break; }; @@ -2723,7 +2743,7 @@ const MapProject = () => { rawResponse: response }), {isRunTraffic: true}) } - })), () => abortRef.current); // wait for completion + })), isStopped); // wait for completion await new Promise(resolve => setTimeout(resolve, 500)); // 1s delay } const now = moment() From 3b2e0e70cf2d2bf29da6fbc170a0bc4a07998bb0 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Wed, 30 Sep 2026 00:04:29 -0400 Subject: [PATCH 10/14] OpenConceptLab/ocl_issues#2849 | Codex pass 9: Stop doesn't wait on the AI step's calls, and a superseded ScispaCy answer leaves the panel alone - The run's AI step no longer waits past Stop for the prompt template or a row's AI call that never answers (a 429 from the template fetch never settled): a stopped run goes on to its autosave and closes its record, and the call finishes in the background. The AI Assistant calls' own retry is OpenConceptLab/ocl_online#206. - A ScispaCy answer from a run a newer run replaced no longer clears the alert or the panel's loading state. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 26 ++++++++++++++++++---- 1 file changed, 22 insertions(+), 4 deletions(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index 75cad71f..672a403c 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -2622,7 +2622,20 @@ const MapProject = () => { setBulkAIAnalysisStartedAt(moment()) let resolvedPromptTemplate try { - resolvedPromptTemplate = await resolvePromptTemplateForInvocation() + // Stop doesn't wait on a call that never answers (ocl_issues#2849); the + // AI Assistant calls' own retries are OpenConceptLab/ocl_online#206. + const outcome = await untilCancelled(resolvePromptTemplateForInvocation(), isStopped) + if(!outcome.settled) { + if(isCurrent()) { + const now = moment() + setBulkAIAnalysisEndedAt(now) + setEndMatchingAt(now) + setLoadingMatches(false) + setIsLoadingInDecisionView(false) + } + return + } + resolvedPromptTemplate = outcome.value } catch (err) { const now = moment() setBulkAIAnalysisEndedAt(now) @@ -2640,7 +2653,8 @@ const MapProject = () => { // starts, so nothing else aborts. if (shouldStopAIStep({quotaExhausted: aiQuotaExhaustedRef.current, failuresInARow: aiFailuresInARowRef.current})) break; - await fetchRecommendation(_rows[index], resolvedPromptTemplate, true); + // A stopped run moves on; the row's call finishes in the background. + await untilCancelled(fetchRecommendation(_rows[index], resolvedPromptTemplate, true), isStopped); } if(!isCurrent()) return @@ -4072,9 +4086,11 @@ const MapProject = () => { const runTicket = autoMatchRunTicketRef.current const isSuperseded = () => isBulk && runTicket !== autoMatchRunTicketRef.current const capacityWait = trackCapacityWait([__row.__index], {algo: SCISPACY_ALGO_ID}) + // A newer run's stages and panel state are its own. const endStopped = () => { - if(!isSuperseded()) - markAlgo(__row.__index, SCISPACY_ALGO_ID, -1) + if(isSuperseded()) + return + markAlgo(__row.__index, SCISPACY_ALGO_ID, -1) setIsLoadingInDecisionView(false) } @@ -4102,6 +4118,8 @@ const MapProject = () => { if(result.ok) { const response = result.response + if(isSuperseded()) + return response // Clear ScispaCy's own warming-up or starting notice, but not an // error another algorithm for this row showed (ocl_online#283). setAlert(prev => (prev?.severity === 'error' ? prev : false)) From eec23c53d5b2aaa933c6668e1525010ab3973321 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Wed, 30 Sep 2026 00:08:57 -0400 Subject: [PATCH 11/14] OpenConceptLab/ocl_issues#2849 | Codex pass 10: a run's pending rerank debounce is dropped once a newer run starts A per-row rerank that a run scheduled (300 ms debounce) and that fired after the user stopped it and started another run passed as the new run's work: a long refusal on it could switch off the new run's reranks. The timer now remembers its run and is dropped if a newer one has started; the new run reranks its own rows. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index 672a403c..6f984eb8 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -4527,8 +4527,14 @@ const MapProject = () => { if(!hasEligiblePending) return if(rerankDebounceRef.current[rowIndex]) clearTimeout(rerankDebounceRef.current[rowIndex]) + // A run's rerank belongs to that run: if a newer run started before the + // timer fired, drop it (the newer run reranks its own rows) rather than + // let it pass as the newer run's (ocl_issues#2849). + const scheduledTicket = autoMatchRunTicketRef.current rerankDebounceRef.current[rowIndex] = setTimeout(() => { delete rerankDebounceRef.current[rowIndex] + if(isRunTraffic && scheduledTicket !== autoMatchRunTicketRef.current) + return // isBulk=false (don't double the setAutoMatched side-effect the explicit // bulk rerank already performs); isRunTraffic carries the run attribution. rerank(rowIndex, false, isRunTraffic) From 75196575aca11614938d600983fceabb1b55e1f4 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Wed, 30 Sep 2026 00:16:11 -0400 Subject: [PATCH 12/14] OpenConceptLab/ocl_issues#2849 | Codex pass 11: a rerank's follow-up runs through the current render's scheduler A rerank's completion fired its follow-up through its own, possibly older, closure: after Stop, an encoder change and a new run, the follow-up reranked with the old encoder and passed as the new run's work. It now goes through scheduleRerankRef, the current render's scheduler. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LxsPK3oiPfG85myNX2VDSj --- src/components/map-projects/MapProject.jsx | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index 6f984eb8..f94f5ee5 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -4460,10 +4460,12 @@ const MapProject = () => { if(inFlightRerankRef.current.get(index) === ownInFlight) inFlightRerankRef.current.delete(index) settleInFlight() - // If new ConceptRows arrived while we were in flight, fire again. + // If new ConceptRows arrived while we were in flight, fire again, + // through the current render's scheduler: this call's closure may hold + // an older encoder or row input (ocl_issues#2849). if(rerankRerunNeededRef.current.has(index)) { rerankRerunNeededRef.current.delete(index) - scheduleRerank(index) + scheduleRerankRef.current(index) } } } From 43649ce8fb5fe37eed9c8d6bd0dffc23aef3ed93 Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Thu, 1 Oct 2026 00:48:07 -0400 Subject: [PATCH 13/14] OpenConceptLab/ocl_issues#2849 | Send capacity_aware on OCL's gated calls, so the server's capacity limit can enforce for this Mapper oclapi2's capacity limit (OpenConceptLab/ocl_online#275) refuses only clients that say they wait out its 429s: "capacity_aware": "true" in X-OCL-Event-Metadata (enforce_for=aware, the default). This Mapper waits them out but never said so, so enforcement could never reach it. It now sends the flag on the calls the limit gates: the bulk and single-row $match (not to a custom algorithm's server), the bridge $match and $rerank. The single-row $match sent no attribution headers before; it now sends them, stamped mapper-ui-manual. Found in the overnight integration test of oclapi2#922 with this PR. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01TyAC9YAn5hP3yGSREkSN8m --- src/components/map-projects/MapProject.jsx | 14 +++++++++----- src/services/__tests__/attribution.test.js | 18 ++++++++++++++++++ src/services/attribution.js | 5 +++++ 3 files changed, 32 insertions(+), 5 deletions(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index f94f5ee5..0a28f9e6 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -1836,11 +1836,13 @@ const MapProject = () => { // call (panel open, single rerun, per-row AI) that overlaps a running // auto-match is correctly stamped 'mapper-ui-manual' — the active-run ref // alone can't distinguish concurrent manual traffic from run traffic. - const attrHeaders = ({rowIndex, rowIndices, batchSize, algorithmId, clientAttemptN, isRunTraffic = false} = {}) => + // capacityAware: the call waits out a capacity 429 (OpenConceptLab/ocl_online#275). Set it on OCL's gated calls + // (semantic or reranked $match, $rerank), never on a custom algorithm's own server. + const attrHeaders = ({rowIndex, rowIndices, batchSize, algorithmId, clientAttemptN, isRunTraffic = false, capacityAware = false} = {}) => buildAttributionHeaders({ runId: isRunTraffic ? (automatchRunRef.current?.id ?? null) : null, projectId: params.projectId, - rowIndex, rowIndices, batchSize, algorithmId, clientAttemptN, + rowIndex, rowIndices, batchSize, algorithmId, clientAttemptN, capacityAware, }) // Create the AutomatchRun system-of-record at run start (oclapi2#876), which @@ -2306,7 +2308,7 @@ const MapProject = () => { payload, (algo.type === 'custom' && algo.url && algo.token) ? algo.token : null, { - headers: attrHeaders({rowIndices: rowIndexes, batchSize: rowBatch.length, algorithmId: algo.id, clientAttemptN: attempt + 1, isRunTraffic: true}), + headers: attrHeaders({rowIndices: rowIndexes, batchSize: rowBatch.length, algorithmId: algo.id, clientAttemptN: attempt + 1, isRunTraffic: true, capacityAware: algo.type !== 'custom'}), query: { includeSearchMeta: true, ...(algo.query_params || {}), @@ -3813,6 +3815,8 @@ const MapProject = () => { payload, (algoDef.type === 'custom' && algoDef.url) ? algoDef.token : null, { + // A custom algorithm's own server gets no new header (it might not allow it in CORS). + ...(algoDef.type === 'custom' && algoDef.url ? {} : {headers: attrHeaders({rowIndex: __row.__index, algorithmId: algoDef.id, capacityAware: true})}), query: { includeSearchMeta: true, includeRetired: isBoolean(_retired) ? _retired : retired, @@ -4370,7 +4374,7 @@ const MapProject = () => { q: query, rows: latestRerankRows.length ? latestRerankRows : rerankRows, ...(encoderModel ? { encoder_model: encoderModel } : {}) - }, null, {headers: attrHeaders({rowIndex: index, algorithmId: 'reranker', isRunTraffic}), handlesThrottle: true, timeout: HEAVY_REQUEST_TIMEOUT_MS}), { + }, null, {headers: attrHeaders({rowIndex: index, algorithmId: 'reranker', isRunTraffic, capacityAware: true}), handlesThrottle: true, timeout: HEAVY_REQUEST_TIMEOUT_MS}), { gate: getCapacityGate(service.URL), isCancelled, onCapacity: noteCapacity, @@ -4769,7 +4773,7 @@ const MapProject = () => { // applied by the premium component (react-bridge-match). Per-row here // (single-row payload) → scalar row_index. isBulk distinguishes a run // call from a manual per-row bridge fetch. - attrHeaders({rowIndex: __row.__index, algorithmId: bridgeAlgoId, isRunTraffic: isBulk}) + attrHeaders({rowIndex: __row.__index, algorithmId: bridgeAlgoId, isRunTraffic: isBulk, capacityAware: true}) ) } catch (err) { failBridgeRow(err?.message || t('unknown_error')) diff --git a/src/services/__tests__/attribution.test.js b/src/services/__tests__/attribution.test.js index e2f12b86..93307355 100644 --- a/src/services/__tests__/attribution.test.js +++ b/src/services/__tests__/attribution.test.js @@ -239,3 +239,21 @@ test('a throttled rerank (-4) makes the run partial; the row still counts as com const r = summarizeRunCompletion({ rowStages, rowIndices: [0, 1], algoIds: ['ocl-search'] }) assert.deepEqual(r, { completed_rows: 2, failed_rows: 0, completion_status: 'partial' }) }) + +test('capacityAware: the capacity_aware flag, as the string "true" (oclapi2 enforce_for=aware)', () => { + const m = meta(buildAttributionHeaders({ runId: 7, projectId: 2, rowIndices: [1, 2], algorithmId: 'ocl-semantic', capacityAware: true })) + assert.equal(m.capacity_aware, 'true') + assert.equal(typeof m.capacity_aware, 'string') + assert.equal(m.automatch_run_id, '7') +}) + +test('capacityAware: left out unless asked for', () => { + assert.equal('capacity_aware' in meta(buildAttributionHeaders({ runId: 7, algorithmId: 'ocl-semantic' })), false) + assert.equal('capacity_aware' in meta(buildAttributionHeaders({ capacityAware: false })), false) +}) + +test('capacityAware on a manual call: mapper-ui-manual source, flag still sent', () => { + const h = buildAttributionHeaders({ projectId: 2, rowIndex: 5, algorithmId: 'ocl-semantic', capacityAware: true }) + assert.equal(h[REQUEST_SOURCE_HEADER], REQUEST_SOURCE.MANUAL) + assert.deepEqual(meta(h), { map_project_id: '2', row_index: '5', algorithm_id: 'ocl-semantic', capacity_aware: 'true' }) +}) diff --git a/src/services/attribution.js b/src/services/attribution.js index ad225d56..0b746523 100644 --- a/src/services/attribution.js +++ b/src/services/attribution.js @@ -47,6 +47,9 @@ const isPresent = value => * @param {string|null} [opts.algorithmId] algorithm_id (semantic / bridge / reranker / …) * @param {number|null} [opts.clientAttemptN] retry attempt number, 1-based * @param {string|null} [opts.source] explicit request_source override + * @param {boolean} [opts.capacityAware] the call waits out a capacity 429: tells oclapi2's capacity limit + * it may refuse it (OpenConceptLab/ocl_online#275). Only for OCL's + * gated calls: semantic or reranked $match, and $rerank * @returns {Object} the two headers */ export const buildAttributionHeaders = ({ @@ -58,6 +61,7 @@ export const buildAttributionHeaders = ({ algorithmId = null, clientAttemptN = null, source = null, + capacityAware = false, } = {}) => { const requestSource = source || (isPresent(runId) ? REQUEST_SOURCE.AUTOMATCH : REQUEST_SOURCE.MANUAL) @@ -78,6 +82,7 @@ export const buildAttributionHeaders = ({ if(isPresent(algorithmId)) meta.algorithm_id = String(algorithmId) if(isPresent(clientAttemptN)) meta.client_attempt_n = String(clientAttemptN) + if(capacityAware) meta.capacity_aware = 'true' return { [REQUEST_SOURCE_HEADER]: requestSource, From 9d1bb52c23341844b2325948a113d2e5c685e82d Mon Sep 17 00:00:00 2001 From: Jonathan Payne Date: Thu, 1 Oct 2026 01:02:40 -0400 Subject: [PATCH 14/14] OpenConceptLab/ocl_issues#2849 | capacity_aware: one tested rule for which $match calls go to OCL (Codex pass 1 on the flag) A custom algorithm without its own URL falls back to OCL's $match, so its bulk calls carry the flag too, as the single-row path already did. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01TyAC9YAn5hP3yGSREkSN8m --- src/components/map-projects/MapProject.jsx | 7 +++---- src/services/__tests__/attribution.test.js | 10 ++++++++++ src/services/attribution.js | 7 +++++++ 3 files changed, 20 insertions(+), 4 deletions(-) diff --git a/src/components/map-projects/MapProject.jsx b/src/components/map-projects/MapProject.jsx index 0a28f9e6..668a79b6 100644 --- a/src/components/map-projects/MapProject.jsx +++ b/src/components/map-projects/MapProject.jsx @@ -70,7 +70,7 @@ import pick from 'lodash/pick' import { OperationsContext } from '../app/LayoutContext'; import APIService, { isTransientNetworkError, retryWithBackoff } from '../../services/APIService'; -import { buildAttributionHeaders, buildConfigSnapshot, summarizeRunCompletion } from '../../services/attribution' +import { buildAttributionHeaders, buildConfigSnapshot, summarizeRunCompletion, matchesOnOCL } from '../../services/attribution' import { CAPACITY_WAIT_CAP_MS, HEAVY_REQUEST_TIMEOUT_MS, LIGHT_REQUEST_TIMEOUT_MS, createCapacityGate, createLimiter, isRetryableError, requestWithCapacityRetry, sleepUnlessCancelled, untilCancelled } from '../../services/capacity' import { highlightTexts, dropVersion, getCurrentUser, hasAuthGroup, hasCapability, getMapperPreview, getNewProjectBlockReason, downloadObject, currentUserToken, refreshCurrentUserCapabilitiesCache } from '../../common/utils'; import { WHITE, SURFACE_COLORS, TEXT_GRAY } from '../../common/colors'; @@ -2308,7 +2308,7 @@ const MapProject = () => { payload, (algo.type === 'custom' && algo.url && algo.token) ? algo.token : null, { - headers: attrHeaders({rowIndices: rowIndexes, batchSize: rowBatch.length, algorithmId: algo.id, clientAttemptN: attempt + 1, isRunTraffic: true, capacityAware: algo.type !== 'custom'}), + headers: attrHeaders({rowIndices: rowIndexes, batchSize: rowBatch.length, algorithmId: algo.id, clientAttemptN: attempt + 1, isRunTraffic: true, capacityAware: matchesOnOCL(algo)}), query: { includeSearchMeta: true, ...(algo.query_params || {}), @@ -3815,8 +3815,7 @@ const MapProject = () => { payload, (algoDef.type === 'custom' && algoDef.url) ? algoDef.token : null, { - // A custom algorithm's own server gets no new header (it might not allow it in CORS). - ...(algoDef.type === 'custom' && algoDef.url ? {} : {headers: attrHeaders({rowIndex: __row.__index, algorithmId: algoDef.id, capacityAware: true})}), + ...(matchesOnOCL(algoDef) ? {headers: attrHeaders({rowIndex: __row.__index, algorithmId: algoDef.id, capacityAware: true})} : {}), query: { includeSearchMeta: true, includeRetired: isBoolean(_retired) ? _retired : retired, diff --git a/src/services/__tests__/attribution.test.js b/src/services/__tests__/attribution.test.js index 93307355..256de2e1 100644 --- a/src/services/__tests__/attribution.test.js +++ b/src/services/__tests__/attribution.test.js @@ -16,6 +16,7 @@ import assert from 'node:assert/strict' import { buildAttributionHeaders, + matchesOnOCL, buildConfigSnapshot, summarizeRunCompletion, REQUEST_SOURCE, @@ -257,3 +258,12 @@ test('capacityAware on a manual call: mapper-ui-manual source, flag still sent', assert.equal(h[REQUEST_SOURCE_HEADER], REQUEST_SOURCE.MANUAL) assert.deepEqual(meta(h), { map_project_id: '2', row_index: '5', algorithm_id: 'ocl-semantic', capacity_aware: 'true' }) }) + +test('matchesOnOCL: built-in algorithms and a custom one without a URL go to OCL; a custom URL does not', () => { + assert.equal(matchesOnOCL({ id: 'ocl-semantic', type: 'ocl-semantic' }), true) + assert.equal(matchesOnOCL({ id: 'ocl-search', type: 'ocl-search' }), true) + assert.equal(matchesOnOCL({ id: 'ocl-bridge', type: 'ocl-bridge' }), true) + assert.equal(matchesOnOCL({ id: 'c', type: 'custom' }), true) + assert.equal(matchesOnOCL({ id: 'c', type: 'custom', url: 'https://example.org/$match/' }), false) + assert.equal(matchesOnOCL(undefined), true) +}) diff --git a/src/services/attribution.js b/src/services/attribution.js index 0b746523..3b7c6852 100644 --- a/src/services/attribution.js +++ b/src/services/attribution.js @@ -31,6 +31,13 @@ export const EVENT_METADATA_HEADER = 'X-OCL-Event-Metadata' const isPresent = value => value !== null && value !== undefined && !(typeof value === 'number' && Number.isNaN(value)) +/** + * Whether an algorithm's $match goes to OCL's API: every algorithm but a custom one with its own URL (which falls + * back to OCL's $match without one, like getMatchAPIService). OCL's calls carry capacity_aware (ocl_online#275); + * a custom algorithm's own server gets no new header, since it might not allow it in CORS. + */ +export const matchesOnOCL = algo => !(algo?.type === 'custom' && algo?.url) + /** * Build the attribution request headers for a single backend call. *