diff --git a/apps/presentation/dashboard/src/features/personal-workspace/capability-configuration-fields.tsx b/apps/presentation/dashboard/src/features/personal-workspace/capability-configuration-fields.tsx index eeea715829..7134ff18b3 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/capability-configuration-fields.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/capability-configuration-fields.tsx @@ -3,7 +3,7 @@ import { useId, type ReactNode } from "react"; import type { CapabilityConfigurationEditor } from "../../data/chat"; import { PeriodicReportScheduleField } from "./periodic-report-schedule-field"; -type FieldCopy = Record; +type FieldCopy = Record }>; type ConfigurationField = CapabilityConfigurationEditor["fields"][number]; type FieldValue = boolean | number | string | string[] | Record | null; type FieldChange = (key: string, value: FieldValue) => void; @@ -40,7 +40,7 @@ function ConfigurationFieldControl({ copy, field, id, onChange, value, timezone {label} ); diff --git a/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts b/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts index 0756239373..25c50ed772 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts +++ b/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts @@ -8,7 +8,7 @@ type LocalizedCopy = Readonly<{ readOnlyReason?: string; }>; -type FieldCopy = Record>; +type FieldCopy = Record }>>; const capabilityCopy: Record> = { en: { @@ -36,7 +36,7 @@ const capabilityCopy: Record> = { }, explore_harness: { displayName: "Explore Harness", - description: "Selects a capability-owned planning and research harness profile for bounded multi-step exploration.", + description: "Plans bounded exploration. Explicit composition replans require a bound experiment or typed result; task completion and execution permissions remain separate.", }, lark_event_inbox: { displayName: "Lark event inbox", @@ -102,7 +102,7 @@ const capabilityCopy: Record> = { }, explore_harness: { displayName: "探索 Harness", - description: "为有界的多步探索选择由能力负责的规划与研究 Harness profile。", + description: "规划有界探索。显式组合的重新规划需绑定实验或类型化结果;任务完成和执行权限单独判断。", }, lark_event_inbox: { displayName: "飞书事件收件箱", @@ -162,6 +162,8 @@ const fieldCopy: Record = { reasoning_effort: { label: "Child reasoning effort", description: "For example max; the host must support this model and effort." }, max_children: { label: "Maximum children", description: "Hard upper bound for concurrently delegated child work." }, profile: { label: "Planner profile", description: "Select one registered Explore Harness profile." }, + composition_mode: { label: "Composition policy", description: "Replan requires an exact experiment successor or typed result; it grants no execution authority.", optionLabels: {disabled: "Off", explicit_only: "Explicit candidates"} }, + composition_scope_id: { label: "Research coverage scope", description: "An opaque scope id required by the explicit-only composition policy." }, profile_preset: { label: "Report profile", description: "Capability-owned report profile, such as weekly-progress." }, wait_for_ci: { label: "Wait for CI", description: "Disable to use local validation without querying or waiting for CI. Merge authority is unchanged." }, review_priority: { label: "Review priority", description: "Choose whether other developers' PRs or the authenticated reviewer's own PRs are ranked first." }, @@ -189,6 +191,8 @@ const fieldCopy: Record = { reasoning_effort: { label: "子 Agent 推理档位", description: "例如 max;宿主须支持所选模型与档位。" }, max_children: { label: "最大子 Agent 数", description: "可同时委派的子任务硬上限。" }, profile: { label: "规划 Profile", description: "选择一个已注册的 Explore Harness profile。" }, + composition_mode: { label: "组合策略", description: "重新规划需要精确的实验后继或类型化结果;它不授予执行权限。", optionLabels: {disabled: "关闭", explicit_only: "仅显式候选"} }, + composition_scope_id: { label: "研究覆盖范围", description: "显式组合策略要求填写不含私有内容的范围标识。" }, profile_preset: { label: "报告 Profile", description: "由该能力管理的报告 profile,例如 weekly-progress。" }, wait_for_ci: { label: "等待 CI", description: "关闭后使用本地验证,不查询或等待 CI;不改变合并权限。" }, review_priority: { label: "审阅优先级", description: "选择先排其他开发者的 PR,还是先排当前已认证审阅者自己的 PR。" }, diff --git a/apps/presentation/dashboard/src/features/personal-workspace/goal-capability-settings.tsx b/apps/presentation/dashboard/src/features/personal-workspace/goal-capability-settings.tsx index d4e48be011..cb7c40dabc 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/goal-capability-settings.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/goal-capability-settings.tsx @@ -1,4 +1,4 @@ -import { useEffect, useMemo, useState } from "react"; +import { useEffect, useLayoutEffect, useMemo, useState } from "react"; import { AlertTriangle, Code2, LoaderCircle, RefreshCw } from "lucide-react"; import { @@ -54,7 +54,9 @@ function useCapabilityMutation({ goalId, onApplied, selected, t }: Readonly<{ ? parseEditableCapabilityJson(selected.configuration_editor, jsonDraft) : null, [selected, jsonDraft]); const jsonValid = editorMode === "guided" || parsedJson !== null; - useEffect(() => { + // Initialize the new capability's draft before it can receive input. A + // post-paint reset can otherwise erase the first toggle after selection. + useLayoutEffect(() => { setEditorMode("guided"); setJsonDraft(""); setMutation({ diff --git a/docs/architecture/rfcs/research-exploration-control-plane-v0.md b/docs/architecture/rfcs/research-exploration-control-plane-v0.md index 2c5bfa54ac..dc8590a138 100644 --- a/docs/architecture/rfcs/research-exploration-control-plane-v0.md +++ b/docs/architecture/rfcs/research-exploration-control-plane-v0.md @@ -150,10 +150,8 @@ milestone status. | Gap | Consequence | |---|---| -| No shared research write-time gate | The cold evidence codec cannot discharge or enforce a live composition obligation. | -| No exact obligation/Todo/result lineage | A research receipt is not proof of an authorized Todo transition or accepted Goal closure. | -| Cold shadow not adopted by hot status/frontier | The existing #3173 projection remains behavior-compatible; canonical research obligations still need M3 integration. | -| No dismissal or deferral contract | Evidence invalidation is visible, but typed candidate retirement and resumption remain unimplemented. | +| Integrated M3 qualification remains | Typed dismissal, blocker waits and exact invalidated-duty retirement are implemented locally; final premerge and maintainer-reviewed integration remain distinct from implementation and live research qualification. | +| Execution attribution is not effect authority | The live replan gate joins exact Todo/experiment/input facts; its receipt is not task-lease/effect authorization or accepted Goal closure. | | Live qualification incomplete | Deterministic and real CLI/file-log tests establish state semantics, not model selection quality or scientific truth; no live Lark sync is qualified by projection tests. | | No promotion evidence for inferred combinations | Shared constraints are not known to be precise enough to trigger obligations. | @@ -232,8 +230,9 @@ execution, and result into one ambiguous relation. This RFC rejects that shape. The research envelope and closure basis have an active CLI caller and the [versioned evidence protocol](../../reference/protocols/research-observation-v0.md). -The action signature, shared write gate and model selection below remain design -targets. The cold shadow does not promote them into current behavior. +The opt-in M3 development path implements exact execution lineage and shared +write gates and bounded retirement transitions. Model selection below +remain design targets; the cold shadow does not activate enforcement. ### 7.1 Compose; do not mutate v0 silently @@ -804,7 +803,7 @@ control-plane failures. | M0 | RFC, current-state inventory, and explicit ownership decision | Maintainer review; no runtime behavior | Accepted design | | M1 | Characterization fixtures plus typed research observation and closure contract in Explore | Deterministic normalization, privacy, compatibility, and negative tests | Implemented evidence/CLI slice; live research qualification remains separate | | M2 | Explicit-only composition candidate, canonical gap projection, and read-only status shadow | No pairwise inference; bounded packet; projection parity | Partial: #3173 legacy quota/successor; canonical binary cold shadow in CLI/Lark projection; hot status adoption and live Lark qualification remain | -| M3 | Goal-frontier obligation, exact Todo/experiment lineage, and shared write-time gate | State/replay matrix and premerge canary pass | Not started | +| M3 | Goal-frontier obligation, exact Todo/experiment lineage, and shared write-time gate | State/replay matrix and premerge canary pass | Local state/replay and standard premerge qualified; maintainer-reviewed integration pending. Scoped policy/editor, shared gates, archive lineage, consumers, dismissal, waits and exact no-spend duty retirement are implemented | | M4 | Bounded multi-candidate cards, `composition_selection_v0`, real model-tool behavior qualification, and repeated live shadow | Model autonomously selects a legal semantic action from the delivered candidate set; selection quality is no worse than the declared fallback; compact receipts only | Not started | | M5 | Shared-constraint candidate ranking in shadow mode | Precision and cost evidence; no automatic trigger | Not started | | M6 | Optional inferred trigger | Explicit maintainer decision and measured promotion thresholds | Deferred | @@ -835,6 +834,33 @@ M3 is the first behavior-changing slice. It should be a separate PR so the obligation and write gate can be reviewed and reverted independently from the evidence schema. +The M3 development boundary joins current same-agent Todo, experiment and input +facts in the typed Explore owner. Goal policy is explicitly scoped and disabled +by default; the existing capability editor and CLI share its configuration +owner. Quota and refresh reuse one live frontier, with the original duty pinned +through scalar rollout fields and both receipt adapters. Real CLI tests reject +unrelated/deferred successors and invalidated writeback, then settle the original +Turn through its exact successor. File/SQLite tests distinguish canonical Todos +from stale display rows; packaged UI checks exercise policy preview, activation, +disable, readback and narrow screens. Native actor/lease and CAS admission remain in force; a fixed IO host +locks the graph while the typed owner qualifies and persists completion evidence. +Real File/SQLite and legacy tests cover missing evidence, direct IPC self-approval, +terminal-verb bypass, retained archive lineage and immutable completion replay. +Agent-scoped status, Explore and existing Lark Summary fields use the same live +facts. Typed candidate dismissal permits scoped terminal retirement; a fresh +canonical blocker and common Todo resume condition defer the gap without closing +it. Real CLI validates exact observed/dismissed/blocked progress source through +the original Turn, with independent File/SQLite lease-safe wait/resume evidence. +Invalidated input/scope/activation produces source-qualified retirement for the +original duty. Common readback retains its guard and historical debit, closes +that Turn without spend, and keeps the current frontier and runnable/paused work +visible. IO assembly stays in existing CLI/refresh composition roots; the shared +gate consumes supplied capability facts and the typed owner remains singular. +The local state/replay matrix and 19 standard risk-selected premerge checks pass. +Two existing scheduler ACK tests fail identically on the unchanged base under the +same local runtime; this is retained as a baseline limitation, not called green. +Maintainer-reviewed integration and independent live qualification remain. + ## 17. Rejected Alternatives ### 17.1 Automatically pair nodes with a shared closure stage diff --git a/docs/architecture/rfcs/research-exploration-control-plane-v0.zh-CN.md b/docs/architecture/rfcs/research-exploration-control-plane-v0.zh-CN.md index a497f608fa..0029a5bb07 100644 --- a/docs/architecture/rfcs/research-exploration-control-plane-v0.zh-CN.md +++ b/docs/architecture/rfcs/research-exploration-control-plane-v0.zh-CN.md @@ -131,10 +131,8 @@ hypothesis”或“新 probe family”。它无法持久表达:A 和 B 都已 | 缺口 | 后果 | |---|---| -| 没有 shared research write-time gate | Cold evidence codec 不能解除或强制 live composition obligation。 | -| 没有精确 obligation/Todo/result lineage | Research receipt 不证明已授权 Todo transition 或已接受 Goal closure。 | -| Cold shadow 未接入 hot status/frontier | 既有 #3173 投影保持行为兼容;canonical research obligation 仍需 M3 集成。 | -| 没有 dismissal/deferral contract | Evidence invalidation 可见,但类型化 candidate retirement/resumption 尚未实现。 | +| M3 集成资格仍未完成 | Typed dismissal、blocker wait 和精确 invalidated-duty retirement 已在本地实现;最终 premerge 与 maintainer review 集成仍独立于实现和真实研究 qualification。 | +| 执行 attribution 不是 effect 权限 | Live replan 门禁连接精确 Todo/experiment/input 事实;receipt 不证明 task-lease/effect 授权或已接受 Goal closure。 | | Live qualification 不完整 | Deterministic 与真实 CLI/file-log 测试证明状态语义,不证明 model selection 质量或科学结论;projection 测试不构成 live Lark sync 资格。 | | inferred combination 没有 promotion evidence | 共享 constraint 的精度还不足以直接触发 obligation。 | @@ -209,8 +207,8 @@ A、B 之间的 `joint_probe` 直连边会把 candidate、execution 和 result Research envelope 与 closure basis 已有真实 CLI caller 和 [版本化证据协议](../../reference/protocols/research-observation-v0.zh-CN.md)。 -下文 action signature、shared write gate 和 model selection 仍为设计目标; -cold shadow 不会将其 promotion 为当前行为。 +Opt-in M3 开发路径已实现精确 execution lineage 和共享 write gate。下文 model +selection 仍为设计目标;有界 retirement 已实现,cold shadow 不会激活门禁。 ### 7.1 组合,而不是静默修改 v0 @@ -742,7 +740,7 @@ rule,以及 model variance 与 control-plane failure 的分离。 | M0 | RFC、current-state inventory 与显式 ownership decision | Maintainer review;无 runtime behavior | 已接受的设计 | | M1 | Characterization fixture,以及 Explore 中的 typed research observation 与 closure contract | Deterministic normalization、privacy、compatibility 与 negative test | Evidence/CLI 切片已实现;真实研究 qualification 独立保留 | | M2 | Explicit-only composition candidate、canonical gap projection 与 read-only status shadow | 不做 pairwise inference;packet 有界;projection parity | 部分实现:#3173 legacy quota/successor;CLI/Lark projection 的 canonical binary cold shadow;hot status adoption 与 live Lark qualification 仍未完成 | -| M3 | Goal-frontier obligation、精确 Todo/experiment lineage 与共享 write-time gate | State/replay matrix 与 premerge canary 通过 | 未开始 | +| M3 | Goal-frontier obligation、精确 Todo/experiment lineage 与共享 write-time gate | State/replay matrix 与 premerge canary 通过 | 本地 state/replay 与 standard premerge 已验证,maintainer review 集成待完成。Scoped policy/editor、共享门禁、archive lineage、消费者、dismissal、wait 与精确无支出 duty retirement 已实现 | | M4 | 有界 multi-candidate card、`composition_selection_v0`、真实 model-tool behavior qualification 与重复 live shadow | 模型从交付 candidate set 中自主选择合法 semantic action;选择质量不劣于 declared fallback;只保留 compact receipt | 未开始 | | M5 | Shared-constraint candidate 在 shadow mode 中排序 | 有 precision/cost evidence;不自动触发 | 未开始 | | M6 | 可选 inferred trigger | 显式 maintainer decision 与量化 promotion threshold | 延后 | @@ -771,6 +769,27 @@ Projection 测试不证明 live remote effect 或模型自主研究行为。交 M3 是第一个 behavior-changing slice。它应单独成 PR,使 obligation 与 write gate 能够独立于 evidence schema 评审和回滚。 +M3 开发边界在 typed Explore owner 中连接当前同一 Agent 的 Todo、experiment +和 input 事实。Goal policy 显式限定范围且默认关闭;既有能力编辑器与 CLI 共享 +配置 owner。Quota 与 refresh 复用同一 live frontier,原 duty 通过 rollout 标量 +字段和两端 receipt adapter 固定。真实 CLI 测试拒绝无关/延期 successor 与失效 +writeback,再通过精确 successor 结算原 Turn。File/SQLite 测试区分 canonical +Todo 与陈旧显示行;打包 UI 验证策略预览、启用、关闭、读回和窄屏。Native actor/lease 与 CAS admission +仍强制执行;固定 IO host 锁定图,typed owner 校验并持久化 completion evidence。 +真实 File/SQLite 与 legacy 测试覆盖缺少证据、direct IPC 自报 approval、terminal +verb 绕过、保留 archive lineage 和不可变 completion 回放。Agent 范围的 status、 +Explore 与既有 Lark Summary 字段使用同一 live fact。Typed candidate dismissal +允许 scoped terminal retirement;新 canonical blocker 与通用 Todo resume condition +使 gap 暂缓,但不关闭。真实 CLI 通过原 Turn 校验 observed/dismissed/blocked 的 +精确 progress source;File/SQLite 独立验证 lease-safe wait/resume。 +输入/scope/activation 失效时,为原 duty 生成 source-qualified retirement。 +通用 readback 保留原 guard 与历史 debit,无支出关闭该 Turn,同时保持当前 +frontier 与 runnable/paused work 可见。IO 装配保留在既有 CLI/refresh composition +root;共享门禁消费传入的 capability fact,typed owner 始终只有一个。 +本地 state/replay matrix 与 19 项 standard 风险验证通过。两个既有 scheduler ACK +测试在相同本地 runtime、未修改 base 上也以相同方式失败;保留为 baseline 限制, +不称为 green。Maintainer review 集成与独立 live qualification 仍未完成。 + ## 17. 被拒绝的替代方案 ### 17.1 自动组合拥有共享 closure stage 的 node diff --git a/docs/reference/handoff-mode.md b/docs/reference/handoff-mode.md index 161df7f5e3..f0c7519c7c 100644 --- a/docs/reference/handoff-mode.md +++ b/docs/reference/handoff-mode.md @@ -207,6 +207,22 @@ permission. The original rejection code and all fences remain unchanged: | `soft_claim`, non-open Todo, acceptance hold or conflicting write scopes | Resolve the reported acquisition blocker. No acquire action is offered. | | Edit changes retained leased work requirements or status | Use the owning lifecycle transition; acquiring another lease cannot authorize the metadata edit. | +The existing `open -> blocked -> open` lifecycle also accepts a typed +prerequisite wait: use `todo update --status blocked --resume-when +todo_done: --reason ''` after the active holder +releases its execution lease. This explicit wait form is newly supported for +native hard-lease Todos; the prior clear-wait pause form is unchanged. A live +lease or bundled execution proof still rejects the transition. Resume with +`--status open --clear-resume-when --reason ''`; neither transition +grants execution authority, and the next execution requires a fresh lease. + +既有 `open -> blocked -> open` lifecycle 也支持 typed 前置任务等待:active +holder 先释放执行 lease,再用 `todo update --status blocked --resume-when +todo_done: --reason ''` 暂停。此显式 wait 形式 +新增支持 native hard-lease Todo;原 clear-wait pause 形式保持不变。有效 lease +或附带执行 proof 仍被拒绝。用 `--status open --clear-resume-when --reason +''` 恢复;两次转换都不授予执行权限,下一次执行仍需新 lease。 + The recovery descriptor uses the standalone `loopx task-lease acquire` command. Combined `todo claim --task-lease-idempotency-key` is restricted to `hard_lease` and is not the recovery route for `legacy`. Acquire uses `--owner`, a **fresh** diff --git a/docs/reference/protocols/research-observation-v0.md b/docs/reference/protocols/research-observation-v0.md index d307ff4bbd..f6b5369162 100644 --- a/docs/reference/protocols/research-observation-v0.md +++ b/docs/reference/protocols/research-observation-v0.md @@ -3,7 +3,11 @@ Explore owns this optional evidence contract. Its first consumer is `loopx explore observe`; `loopx explore summary` and the existing Lark Explore node summary render the same derived facts. This is the M1 evidence substrate -and M2 **read-only shadow**, not M3 obligation/writeback enforcement. +and M2 **read-only shadow**. The M3 development boundary adds optional execution +attribution, an explicit-only live replan gate, and legacy/native File/SQLite +Todo closeout checks, source-qualified duty retirement, lease-fenced resumption, +and shared status/Explore presentation. Maintainer-reviewed integration and live +model/scientific or remote Lark qualification remain open. ## Record and read back @@ -83,7 +87,7 @@ optional `research_observation` on an append-only node revision. Existing events without that field preserve their projection shape and require no research runtime operation. Older readers that validate node fields strictly must upgrade before reading a log containing the optional envelope. Existing -`#3173` composition/quota behavior is unchanged. No research policy, scheduler, +`#3173` composition/quota behavior is unchanged without the new policy. No research policy, scheduler, claim, lease, quota, generic settlement or Goal acceptance is enabled by this command. @@ -112,9 +116,192 @@ the node writer and batch writer enforce attribution for typed envelopes. The cold projection returns total counts and at most three cards, with explicit `projected_count` and `omitted_count`. It makes no ranking-quality claim and does not add a full candidate list to quota packets. Inspect the canonical -node observations to investigate omitted candidates; no obligations are lost -because this slice creates none. Dismissal, deferral, observation-to-Todo -lineage, shared write gates and hot status adoption remain M3 work. +node observations to investigate omitted candidates. Cold cards do not select +obligations. The live policy below uses the full internal candidate set; +compaction cannot erase an enforceable gap. + +## Execution attribution + +An experiment may additionally record `execution_lineage`: + +```json +{ + "schema_version": "research_execution_lineage_v0", + "goal_id": "research-demo", + "gap_id": "research-composition-0123456789abcdef", + "replan_obligation_id": "replan-0123456789abcdef", + "successor_todo_id": "todo_joint", + "agent_id": "research-agent" +} +``` + +Use the actual gap, obligation and Todo identities supplied by the current +work contract; the example tokens do not establish an obligation. Include this +object in the experiment envelope and issue `explore observe --agent-id` with +the same actor. `progress.work_item_id` must equal `successor_todo_id`. +The source registry must contain that Goal. The writer reads the exact Todo +through the existing Todo reader, including promoted canonical authority when +configured; it does not accept a caller-supplied Todo snapshot. + +A new write requires a same-agent, currently runnable `advancement_task` with +`action_kind=joint_probe`, the exact `replan_obligation_id`, and exactly this +experiment in `explore_result_node_refs`. A declared `target_key` must also +equal the experiment id. Deferred, blocked, rejected by the existing acceptance +guard, archived, executor-excluded, other-agent or unrelated work cannot supply +execution attribution. The experiment must belong to this pending binary gap +and match its current input observations. An unchanged/read/ACK observation or +an execution result without evidence is rejected. + +The fingerprint includes this object, research schema, experiment identity, +input fingerprints and generic progress/evidence. Node and batch writers +reject new execution-lineage writes; use `explore observe` for its task read. +Exact historical replay still writes nothing, even after task or input changes. +It does not refresh current task authority. Presentation compaction does not +limit the internal candidate set used for attribution. + +This is a task snapshot attribution check under the Explore log lock, not an +atomic task lease/effect authorization or a Todo/Goal completion receipt. +Any subsequent shared settlement must independently validate its current +authority and research evidence. Envelopes without `execution_lineage` retain +their diagnostic behavior; they cannot be promoted into execution lineage by +reading or replaying them. The live replan gate below adopts these facts; +the same typed rule also guards current legacy and native File/SQLite Todo closeout. + +Completed Todo records retain evidence lineage after archival or claim clearing. +The projection reads retained canonical history rather than a compact active +Todo list. An archived open/deferred row cannot schedule or attribute a new +observation; missing or mismatched history and invalidated inputs still reject +current coverage. Historical evidence grants no current claim, lease or execution +authority. + +## Explicit-only live replan gate (M3 development) + +Preview and apply the policy through the existing Goal configuration owner: + +```sh +loopx configure-goal --goal-id research-demo --explore-harness-enabled \ + --explore-composition-mode explicit_only --explore-composition-scope-id joint-scope +loopx configure-goal --goal-id research-demo --explore-harness-enabled \ + --explore-composition-mode explicit_only --explore-composition-scope-id joint-scope --execute +loopx quota should-run --goal-id research-demo --agent-id research-agent --turn-instance-id research-turn +``` + +The Goal capability editor exposes the same activation, policy and scope fields +through its existing revision-checked preview/apply API. Both harness activation +and `composition_mode=explicit_only` are required. The opaque +`composition_scope_id` identifies experiment coverage; it does not replace the +Goal vision or grant claim, lease, effect, provider or external-execution rights. + +Quota and `refresh-state` read the same live Explore/Todo frontier. A compact +capability guard pins the selected gap and input/policy revision to the original +Turn. The existing common replan owner computes its obligation id. User/handoff +gates, runnable work, vision acceptance and ordinary succession/review duties +retain priority before this capability gap enters monitor fallback. + +Only an exact same-agent runnable `joint_probe` successor with the current +obligation and one current binary experiment suppresses duplicate planning. +`todo add --replan-obligation-id` rejects unrelated or deferred successors. +Scheduling does not observe the gap. A terminal experiment result must include +execution lineage, current input fingerprints and the configured coverage scope +before it becomes live observed evidence. Its complete generic progress and +research fingerprint bind writeback; omitted semantics, unrelated progress, +reads/ACKs, stale inputs and replay cannot discharge the selected duty. + +The supported replan exits are a new runnable experiment successor, its typed +observed result, an evidence-backed candidate dismissal or a fresh exact blocker +with a typed resume condition. A negative scientific conclusion is valid evidence. +Generic blocked/terminal claims do not certify research closure. Current legacy +and native File/SQLite closeout require the task's exact terminal experiment or +dismissal evidence. An invalidated admitted duty uses the lifecycle retirement +below; a Todo becoming done never certifies Goal closure. + +An observation may include `composition_resolution` with schema +`research_composition_resolution_v0`. It requires execution lineage and nonempty +`evidence_ids` attributable to the same progress observation: + +- `disposition=dismissed` requires coverage-backed `no_followup`, a matching + `no_followup` closure basis, an experiment node in `dead_end` and typed `basis` + (`duplicate`, `invalid`, `unsafe` or `outside_scope`). It counts as `dismissed`, + never an observed experiment. The exact evidence permits normal Todo completion + or supersession without pretending the experiment ran. +- `disposition=deferred` requires `blocked` progress and its canonical + `blocker_id`; it cannot assert complete terminal coverage. The linked experiment + is blocked, its current Todo is blocked/deferred with + `resume_when=todo_done:`, and the same-agent active blocker Todo + names that experiment Todo in `unblocks_todo_id`. The common Todo resume owner + must prove the prerequisite is still pending. Missing, unrelated, archived or + already-resolved blockers cannot suppress the gap. Reusing a previously claimed + blocker cannot provide fresh writeback progress. + +These are caller-declared evidence dispositions, not independent scientific +verification or execution permission. Resolving the prerequisite exposes the +pending gap; reopening the task and experiment makes it scheduled, not observed. +Native hard-lease callers must release their active lease before the existing +blocked lifecycle accepts a typed wait. Resume clears the wait and still requires +a fresh execution lease. Arbitrary deferred successors remain rejected. + +When refreshing a replan-bound Turn from a canonical experiment observation, +use `--progress-work-item-id` for the observation's exact source Todo and repeat +its complete generic progress fields. This selector does not rebind settlement +away from the original obligation. The shared gate still verifies the actual +Todo, current inputs, scope, evidence and fingerprint; changed or incomplete +observations are rejected. A new concrete blocker is an explicit replan progress +exit; it does not use the Todo-bound `outcome_gap` no-spend closeout contract. + +The native transaction evaluates capability eligibility against its actual +provider Todo after actor/lease admission and before caller validation effects. +A fixed source-selected Python adapter holds the canonical Explore log lock +through the native CAS. It only reads the graph; TypeScript decides whether +the observation, task and input lineage match. The model cannot supply an +approval flag, snapshot or executable to bypass this guard. Accepted evidence +is retained in the terminal operation receipt. Replaying that receipt reports +history even after later input invalidation, without rewriting state or evidence. +Completing through a different terminal verb cannot avoid the same check; +typed candidate dismissal above is a legal retirement without experimental outcome evidence. + +The writer holds the Explore log lock while qualifying and persisting its +research writeback. The persisted guard uses bounded scalar rollout fields; +both TS and Python receipt adapters retain the same selected facts. Public +quota cards omit internal task joins, transition candidates and full result +observations, with counts for omitted display cards. + +`status --goal-id research-demo --agent-id research-agent` displays these same +live counts and bounded gaps in `project_asset.bounded_research_frontier`. +`explore summary --goal-id research-demo --agent-id research-agent` exposes +`research_execution_frontier` and annotates existing node summaries. The same +summaries feed Lark projection fields; `feishu-sync` and `feishu-card` accept +the same Agent selector. A live-policy Goal with multiple registered Agents +requires that explicit selector; a single Agent is selected automatically for +presentation only. These read paths do not start a Turn, mutate Todo state or +authorize remote writes. Inactive policy preserves the existing projections. + +Disable only the new policy while retaining the existing planner: + +```sh +loopx configure-goal --goal-id research-demo --explore-composition-mode disabled --execute +``` + +Removing the policy fields also restores the existing behavior. Changing scope, +inputs or activation cannot retroactively erase an admitted Turn's guard; stale +writeback remains rejected. Its rejection supplies an exact `retirement_contract` +when current source facts prove invalidation. Reuse its blocked progress fields +with the original `--replan-obligation-id` and `--turn-instance-id`, and set +`--delivery-outcome outcome_gap`. The capability owner revalidates the current +source and generates `capability_obligation_retirement_v0`; caller approval flags +or arbitrary blocker/evidence ids cannot provide this proof. Runnable Todos bound +to that duty prevent retirement until their owning lifecycle pauses them. +Invalid, missing or unavailable evidence sources cannot stand in for invalidation. + +The common settlement owner verifies the original guard, durable writeback, +current revision and exact progress fingerprint, then returns +`closeout_kind=capability_duty_retired_no_spend`. It closes only that admitted +Turn. No Todo or Goal is completed, and no quota slot is spent; already committed +debits remain ordinary historical debits. Old-Turn reentry and spend requests +return the same no-debit closeout without appending accounting. The next Turn +reassesses the current frontier, including remaining acceptance and paused work. +Keep the original receipt and evidence rather than deleting them to manufacture +settlement. No policy setting authorizes Goal +termination, live model qualification, benchmark launch, deployment or release. ## Disable and authority boundary diff --git a/docs/reference/protocols/research-observation-v0.zh-CN.md b/docs/reference/protocols/research-observation-v0.zh-CN.md index 37a0664236..a22fb2279f 100644 --- a/docs/reference/protocols/research-observation-v0.zh-CN.md +++ b/docs/reference/protocols/research-observation-v0.zh-CN.md @@ -2,7 +2,10 @@ Explore 拥有这一可选证据契约。第一个写入入口是 `loopx explore observe`; `loopx explore summary` 与现有 Lark Explore 节点摘要显示同一派生事实。 -本批交付 M1 证据基础和 M2 **只读 shadow**,不实现 M3 的 obligation/writeback 门禁。 +本批交付 M1 证据基础和 M2 **只读 shadow**。M3 开发边界增加可选执行 attribution +与 explicit-only live replan 门禁,以及 legacy/native File/SQLite Todo closeout +校验、来源限定的义务退役、lease-fenced 恢复和共享 status/Explore 展示。 +Maintainer review 集成与现场模型/科研、远端 Lark qualification 仍待完成。 ## 录入与读回 @@ -74,7 +77,7 @@ Candidate 必须 `basis=explicit`,引用不同的已知 target,且 evidence 接受的 envelope 获得内容 fingerprint,作为 append-only 节点 revision 的可选 `research_observation` 存储。没有该字段的旧 event 保持 projection 形状,无需调用 research runtime。严格校验节点字段的旧 reader 必须升级后才能读取含新 envelope -的日志。既有 `#3173` composition/quota 行为不变;此命令不启用 research policy、 +的日志。没有新策略时,既有 `#3173` composition/quota 行为不变;此命令不启用 research policy、 scheduler、claim、lease、quota、generic settlement 或 Goal acceptance。 ## 结果、失效与回放 @@ -96,9 +99,162 @@ writer 和 batch writer 同样校验 typed envelope 的 attribution。 Cold projection 返回总数和至多三张 card,并提供 `projected_count`、`omitted_count`。 不声明排序质量,不把完整 candidate 列表加入 quota packet。遗漏项可从 canonical -node observation 检查;本批不创建 obligation,因此不会丢失义务。Dismissal、 -deferral、observation/Todo lineage、shared write gate 和 hot status adoption -仍属 M3。 +node observation 检查。Cold card 不选择 obligation;下述 live policy 使用完整 +内部候选集合,显示截断不能抹去 enforceable gap。 + +## 执行 attribution + +实验 envelope 可额外包含 `execution_lineage`: + +```json +{ + "schema_version": "research_execution_lineage_v0", + "goal_id": "research-demo", + "gap_id": "research-composition-0123456789abcdef", + "replan_obligation_id": "replan-0123456789abcdef", + "successor_todo_id": "todo_joint", + "agent_id": "research-agent" +} +``` + +使用当前工作契约提供的真实 gap、obligation 和 Todo 身份;示例 token 不产生 +义务。将该对象放入 experiment envelope,并用同一 actor 发出 +`explore observe --agent-id`。`progress.work_item_id` 必须等于 +`successor_todo_id`。源 registry 必须包含该 Goal。Writer 通过既有 Todo reader +读取精确任务,包括已配置的 promoted canonical authority,不接受调用者自报的 +Todo 快照。 + +新写入要求当前可执行、归属同一 Agent 的 `advancement_task`,其 +`action_kind=joint_probe`、`replan_obligation_id` 精确匹配,且 +`explore_result_node_refs` 仅包含本实验。声明的 `target_key` 也必须等于实验 id。 +延期、阻塞、被既有 acceptance guard 拒绝、归档、executor 被排除、其他 Agent +或无关任务均不能提供执行 attribution。实验必须对应这个 pending 二元 gap, +精确匹配当前输入 observation。Unchanged/read/ACK 或无 evidence 的结果被拒绝。 + +Fingerprint 包含该对象、research schema、实验身份、输入 fingerprint 和 generic +progress/evidence。Node 和 batch writer 拒绝新的 execution-lineage 写入,须使用 +`explore observe` 的任务读取路径。精确历史回放始终不写入,即使任务或输入已经 +改变,也不刷新当前任务权限。显示 card 的截断不限制内部 attribution 候选集合。 + +这是 Explore 日志锁内的任务快照 attribution 校验,不是原子的任务 lease/effect +授权,也不是 Todo/Goal 完成收据。后续 shared settlement 必须独立校验当前权限和 +研究证据。不带 `execution_lineage` 的 envelope 保持诊断行为,不能经读取或回放 +升级为执行 lineage。下述 live replan 门禁采用这些事实;相同 typed rule +也控制当前 legacy/native File/SQLite Todo closeout。 + +## Explicit-only live replan 门禁(M3 开发中) + +通过既有 Goal 配置 owner 预览、应用策略: + +```sh +loopx configure-goal --goal-id research-demo --explore-harness-enabled \ + --explore-composition-mode explicit_only --explore-composition-scope-id joint-scope +loopx configure-goal --goal-id research-demo --explore-harness-enabled \ + --explore-composition-mode explicit_only --explore-composition-scope-id joint-scope --execute +loopx quota should-run --goal-id research-demo --agent-id research-agent --turn-instance-id research-turn +``` + +Goal 能力编辑器通过既有 revision-checked preview/apply API 展示相同的启用、策略 +与范围字段。Harness 启用与 `composition_mode=explicit_only` 必须同时满足。 +不含私有内容的 `composition_scope_id` 指定实验覆盖范围;不替代 Goal vision, +不授予 claim、lease、effect、provider 或外部执行权限。 + +Quota 与 `refresh-state` 读取同一 live Explore/Todo frontier。Compact capability +guard 将所选 gap 与 input/policy revision 固定在原 Turn;既有 common replan +owner 计算 obligation id。User/handoff gate、可执行工作、vision acceptance 和 +普通 succession/review duty 保持更高优先级,之后 capability gap 才进入 monitor fallback。 + +仅当前同一 Agent、精确 obligation、单个当前二元实验的可执行 `joint_probe` +successor 能抑制重复规划。`todo add --replan-obligation-id` 拒绝无关或延期的 +successor。安排工作不等于 gap 已 observed。Terminal 实验结果须包含 execution +lineage、当前 input fingerprint 和配置的 coverage scope,才能成为 live observed +evidence。完整 generic progress 与 research fingerprint 绑定 writeback;丢失语义、 +无关 progress、read/ACK、过期输入及回放不能解除所选义务。 + +此边界支持可执行实验 successor、其 typed observed result、有证据的 candidate +dismissal 或带 typed resume condition 的新精确 blocker。科学结论为负也是有效 +证据。Generic blocked/terminal claim 不证明研究 closure。当前 legacy/native +File/SQLite closeout 需要该任务的精确 terminal 实验或 dismissal 证据;失效的已 +准入 duty 使用下述 lifecycle retirement。Todo done 始终不证明 Goal closure。 + +Observation 可含 schema 为 `research_composition_resolution_v0` 的 +`composition_resolution`。它需要 execution lineage 和非空 `evidence_ids`, +且证据必须归属于同一 progress observation: + +- `disposition=dismissed` 要求 coverage-backed `no_followup`、匹配的 + `no_followup` closure basis、`dead_end` 实验节点与 typed `basis` + (`duplicate`、`invalid`、`unsafe` 或 `outside_scope`)。它计为 `dismissed`, + 不计为 observed experiment。精确证据允许正常 Todo completion 或 supersession, + 不暗示实验已运行。 +- `disposition=deferred` 要求 `blocked` progress 与 canonical `blocker_id`, + 不得断言完整 terminal coverage。实验节点为 blocked;当前实验 Todo 为 + blocked/deferred,声明 `resume_when=todo_done:`;同一 Agent 的 + 活动 blocker Todo 通过 `unblocks_todo_id` 指向实验 Todo。通用 Todo resume + owner 必须证明该前置任务仍 pending。缺失、无关、归档或已解决的 blocker + 不能压制 gap;已在 writeback 中使用过的 blocker 不能再作为新进展。 + +这些是调用者声明的证据 disposition,不是独立科学验证或执行许可。前置任务 +解决后 gap 重新 pending;重新打开任务和实验仅使其 scheduled,不是 observed。 +Native hard-lease 调用者必须先释放 active lease,既有 blocked lifecycle 才接受 +typed wait。恢复清除 wait 后仍需新的执行 lease;任意 deferred successor 仍被拒绝。 + +从 canonical experiment observation 写回 replan-bound Turn 时,用 +`--progress-work-item-id` 指定其精确源 Todo,并重述完整 generic progress 字段。 +它不把 settlement 从原 obligation 重新绑定到其他工作。共享门禁仍校验真实 +Todo、当前输入、scope、evidence 与 fingerprint;修改或不完整 observation 被拒绝。 +新 concrete blocker 是显式 replan progress exit,不使用 Todo-bound +`outcome_gap` 无支出 closeout 契约。 + +Native 事务在 actor/lease admission 后、caller validation effect 前,使用实际 +provider Todo 校验能力资格。固定的 source-selected Python adapter 持有 canonical +Explore 日志锁直到 native CAS 返回;它只读取图,observation/task/input lineage +是否合法由 TypeScript 判断。模型不能自报 approval flag、snapshot 或 executable +绕过门禁。接受的证据保留在 terminal operation receipt 中。后续输入失效后,精确 +receipt 回放仍只报告历史,不重新写状态或证据。换用其他 terminal verb 也不能避开 +相同校验;上述 typed candidate dismissal 是没有实验结果证据时的合法 retirement。 + +Writer 在校验、持久化 research writeback 时持有 Explore 日志锁。持久化 guard +使用有界 rollout 标量字段;TS 与 Python receipt adapter 保留相同所选事实。 +Public quota card 不携带内部 task join、transition candidate 或完整 result +observation,并提供遗漏 card 的数量。 + +已完成 Todo 在归档或清除 claim 后保留证据 lineage。投影读取 canonical 保留 +历史,而不是压缩的活动 Todo 列表。归档的 open/deferred 行不能调度或提供新 +observation attribution;缺失或不匹配的历史、失效输入仍不能证明当前覆盖。 +历史证据不授予当前 claim、lease 或执行权限。 + +`status --goal-id research-demo --agent-id research-agent` 在 +`project_asset.bounded_research_frontier` 展示同一 live count 与有界 gap。 +`explore summary --goal-id research-demo --agent-id research-agent` 提供 +`research_execution_frontier`,并注释既有 node summary;相同 summary 进入 Lark +投影字段。`feishu-sync` 与 `feishu-card` 支持同一 Agent selector。多个已注册 +Agent 的 live-policy Goal 需要显式 selector;单个 Agent 可自动用于展示。 +这些读取不开始 Turn、不修改 Todo、不授予远程写权限;未激活策略保留原投影。 + +只关闭新策略、保留既有 planner: + +```sh +loopx configure-goal --goal-id research-demo --explore-composition-mode disabled --execute +``` + +移除策略字段也恢复既有行为。范围、输入或激活变化不能追溯抹去已准入 Turn 的 +guard;过期 writeback 仍被拒绝。当前 source fact 证明 invalidation 时,拒绝结果 +提供精确 `retirement_contract`。在原 `--replan-obligation-id` 和 +`--turn-instance-id` 下重用其 blocked progress 字段,并指定 +`--delivery-outcome outcome_gap`。Capability owner 重新校验当前源并生成 +`capability_obligation_retirement_v0`;调用者自报 approval flag 或任意 +blocker/evidence id 不能代替此证明。仍有绑定该 duty 的 runnable Todo 时不能 +retire,须先通过该任务自己的 lifecycle 暂停。无效、缺失或不可用的 evidence +source 不能冒充 invalidation。 + +通用 settlement owner 校验原 guard、durable writeback、当前 revision 与精确 +progress fingerprint,返回 `closeout_kind=capability_duty_retired_no_spend`。 +它仅关闭该已准入 Turn,不完成 Todo/Goal,也不花费 quota;已提交 debit 始终 +保留为普通历史 debit。旧 Turn reentry 与 spend 请求复用同一无支出 closeout, +不追加记账。下一轮重新判断当前 frontier,保留未完成 acceptance 与暂停任务。 +保留原 receipt 与证据,不通过删除它们制造结算。 +任何策略设置都不授权 Goal termination、真实模型 qualification、benchmark launch、 +部署或发布。 ## 停用与权限边界 diff --git a/examples/control_plane/interaction-contract-state-machine-smoke.py b/examples/control_plane/interaction-contract-state-machine-smoke.py index 7e50552b84..442f15a2ea 100644 --- a/examples/control_plane/interaction-contract-state-machine-smoke.py +++ b/examples/control_plane/interaction-contract-state-machine-smoke.py @@ -602,7 +602,8 @@ def assert_required_reads_are_mirrored_into_execution_channels() -> None: "command": " loopx evidence-log --goal-id interaction-state-machine-goal ", } ] - # Reads are mirrored losslessly; the transport does not rewrite commands. + # Execution channels retain the admitted command verbatim. Trimming belongs + # to display compaction and can change quoted arguments. payload["required_reads"] = expected payload = finalize(payload) contract = payload["interaction_contract"] diff --git a/examples/personal-workspace-browser/capability-scope.mjs b/examples/personal-workspace-browser/capability-scope.mjs index f90876ba6a..ed6f1bae19 100644 --- a/examples/personal-workspace-browser/capability-scope.mjs +++ b/examples/personal-workspace-browser/capability-scope.mjs @@ -50,6 +50,50 @@ export const capabilityScopeScenario = { await page.getByRole("button", { name: "预览变更", exact: true }).click(); await page.getByText("锁定 revision 的变更预览", { exact: true }).waitFor(); if (api.goalConfigurationRequests.at(-1)?.goal_id !== "research-monitor") throw new Error("Changed Goal was not used by preview"); + await catalog.getByRole("button", { name: /探索 Harness/ }).click(); + await page.getByRole("switch", { name: /^启用/u }).check(); + const composition = page.getByRole("combobox", { name: /^组合策略/u }); + await composition.selectOption("explicit_only"); + await page.getByLabel(/^研究覆盖范围/u).fill("joint-scope"); + await page.getByRole("button", { name: "预览变更", exact: true }).click(); + await page.getByText("锁定 revision 的变更预览", { exact: true }).waitFor(); + const preview = api.goalConfigurationRequests.at(-1); + if (preview?.goal_id !== "research-monitor" || preview?.configuration?.composition_mode !== "explicit_only" + || preview?.configuration?.composition_scope_id !== "joint-scope" || preview?.configuration?.enabled !== true) throw new Error("Composition preview lost its Goal, activation or typed scope"); + await composition.selectOption("disabled"); + if (await page.locator(".personal-capability-preview").count()) throw new Error("Composition edit retained a stale preview"); + await composition.selectOption("explicit_only"); + await page.getByRole("button", { name: "预览变更", exact: true }).click(); + await page.getByText("锁定 revision 的变更预览", { exact: true }).waitFor(); + await Promise.all([ + page.waitForResponse((response) => new URL(response.url()).pathname === "/api/chat/goal-configuration" && response.request().method() === "GET"), + page.getByRole("button", { name: "应用此预览", exact: true }).click(), + ]); + await catalog.getByRole("button", { name: /探索 Harness/ }).click(); + if (await composition.inputValue() !== "explicit_only" || await page.getByLabel(/^研究覆盖范围/u).inputValue() !== "joint-scope" + || !await page.getByRole("switch", { name: /^启用/u }).isChecked()) throw new Error("Applied composition configuration did not read back from its Goal"); + await composition.selectOption("disabled"); + await page.getByRole("button", { name: "预览变更", exact: true }).click(); + await page.getByText("锁定 revision 的变更预览", { exact: true }).waitFor(); + await Promise.all([ + page.waitForResponse((response) => new URL(response.url()).pathname === "/api/chat/goal-configuration" && response.request().method() === "GET"), + page.getByRole("button", { name: "应用此预览", exact: true }).click(), + ]); + await catalog.getByRole("button", { name: /探索 Harness/ }).click(); + if (await composition.inputValue() !== "disabled" || await page.getByLabel(/^研究覆盖范围/u).inputValue() !== "joint-scope" + || !await page.getByRole("switch", { name: /^启用/u }).isChecked()) throw new Error("Disabling composition changed its Goal activation or scope"); + const applied = api.goalConfigurationRequests.filter((request) => request.phase === "apply"); + if (applied.length !== 2 || applied.some((request) => request.goal_id !== "research-monitor" + || request.capability_id !== "explore_harness" || request.expected_plan_revision !== "sha256:goal-plan-explore_harness") + || applied[0].configuration.composition_mode !== "explicit_only" + || applied[1].configuration.composition_mode !== "disabled") throw new Error("Composition apply lost its locked revision or target"); + await page.screenshot({ path: resolve(outputDir, "capability-composition-desktop.png"), animations: "disabled" }); + await page.setViewportSize({ width: 390, height: 844 }); + await page.getByLabel(/^研究覆盖范围/u).scrollIntoViewIfNeeded(); + await page.getByLabel(/^研究覆盖范围/u).focus(); + await page.screenshot({ path: resolve(outputDir, "capability-composition-mobile.png"), animations: "disabled" }); + if (await page.evaluate(() => document.documentElement.scrollWidth > innerWidth + 1)) throw new Error("Composition controls overflow the narrow viewport"); + await page.setViewportSize({ width: 1512, height: 982 }); await defaults.check(); await page.getByRole("navigation", { name: "机器能力目录" }).waitFor(); if (await page.locator(".personal-capability-preview").count()) throw new Error("A Goal preview crossed into device defaults"); @@ -57,7 +101,7 @@ export const capabilityScopeScenario = { await page.getByRole("heading", { level: 2, name: "周期报告", exact: true }).waitFor(); if (await page.locator(".personal-capability-preview").count()) throw new Error("Returning revived a discarded preview"); if (!reads.includes("product-release") || !reads.includes("research-monitor")) throw new Error("Target changes did not read their own configuration"); - if (api.goalConfigurationRequests.some((r) => r.phase === "apply") || api.machineConfigurationRequests.some((r) => r.phase === "apply")) throw new Error("Navigation wrote configuration"); + if (api.goalConfigurationRequests.filter((r) => r.phase === "apply").length !== 2 || api.machineConfigurationRequests.some((r) => r.phase === "apply")) throw new Error("Navigation wrote configuration"); await page.screenshot({ path: resolve(outputDir, "capability-scope-goal.png"), animations: "disabled" }); await page.setViewportSize({ width: 390, height: 844 }); await target.focus(); @@ -70,6 +114,9 @@ export const capabilityScopeScenario = { if (!await goalScope.isChecked() || await target.inputValue() !== "product-release") throw new Error("Goal settings entry lost its target"); if (context.errors.length) throw new Error(context.errors.join(" | ")); return { coverageEntries: await context.close(), note: "One capability destination; explicit device/Goal scope; fresh target reads; no cross-scope drafts or writes; Goal entry preserved; desktop/mobile verified." }; - } catch (error) { await context.close(); throw error; } + } catch (error) { + await page.screenshot({ path: resolve(outputDir, "capability-scope-failure.png"), animations: "disabled" }); + await context.close(); throw error; + } }, }; diff --git a/examples/personal-workspace-browser/fixture.mjs b/examples/personal-workspace-browser/fixture.mjs index ca1cf8c2c7..8e56171d1b 100644 --- a/examples/personal-workspace-browser/fixture.mjs +++ b/examples/personal-workspace-browser/fixture.mjs @@ -172,6 +172,8 @@ export function goalCapabilityCatalog(multiSubagentConfiguration) { fields: [ { key: "enabled", label: "Enabled", description: "", input_kind: "boolean", required: false }, { key: "profile", label: "Planner profile", description: "", input_kind: "select", required: false, options: ["generic"] }, + { key: "composition_mode", label: "Composition policy", description: "Replan accepts an exact experiment successor or typed result.", input_kind: "select", required: false, options: ["disabled", "explicit_only"] }, + { key: "composition_scope_id", label: "Research coverage scope", description: "An opaque coverage scope.", input_kind: "text", required: false, nullable: true }, ], }), goalCapability({ capabilityId: "lark_kanban_heartbeat_sync", displayName: "Lark Kanban heartbeat sync" }), @@ -368,7 +370,7 @@ function filterStatusFixtureToScope(fixture, matchesScope) { export async function installApi(page, { goalSubagentConfigurationEnabled = true, initialActionProposals = [], managerChannelBinding = null, progressiveWorkspace = false, runtimeAgents = null } = {}) { let turnCounter = 0; - const runtime = page.__loopxRuntime ??= { actionProposals: new Map(), goalSubagentConfigurations: new Map(), larkConnections: [], messages: new Map(), sessions: new Map(), turnMessages: new Map() }; + const runtime = page.__loopxRuntime ??= { actionProposals: new Map(), goalSubagentConfigurations: new Map(), goalExploreConfigurations: new Map(), larkConnections: [], messages: new Map(), sessions: new Map(), turnMessages: new Map() }; const actionProposals = runtime.actionProposals; const sessions = runtime.sessions; const messages = runtime.messages; @@ -1224,7 +1226,8 @@ export async function installApi(page, { goalSubagentConfigurationEnabled = true machine_default_present: true, effective_revision: "sha256:periodic-effective", }, - }) : capability)], + }) : capability.capability_id === "explore_harness" && runtime.goalExploreConfigurations.has(goalId) + ? { ...capability, current: runtime.goalExploreConfigurations.get(goalId) } : capability)], }, }, status: 200 }); return; @@ -1245,6 +1248,17 @@ export async function installApi(page, { goalSubagentConfigurationEnabled = true if (url.pathname === "/api/chat/goal-configuration/apply" && request.method() === "POST") { const body = request.postDataJSON(); state.goalConfigurationRequests.push({ phase: "apply", ...body }); + if (body.capability_id === "explore_harness") { + runtime.goalExploreConfigurations.set(body.goal_id, body.configuration); + await route.fulfill({ contentType: "application/json", json: { + ok: true, schema_version: "goal_configuration_transaction_v0", status: "applied", + goal_id: body.goal_id, capability_id: body.capability_id, + plan_revision: body.expected_plan_revision, applied_revision: "sha256:goal-explore-applied", + readback_verified: true, changed_fields: ["explore_harness"], goal_configuration: body.configuration, + capability_catalog: { schema_version: "capability_configuration_catalog_v0", capabilities: [] }, + }, status: 200 }); + return; + } if (body.capability_id === "multi_subagent") { runtime.goalSubagentConfigurations.set(body.goal_id, body.configuration.enabled ? { mode: "multi_subagent", diff --git a/loopx/capabilities/configuration_ui.py b/loopx/capabilities/configuration_ui.py index c568294335..642c55e0b2 100644 --- a/loopx/capabilities/configuration_ui.py +++ b/loopx/capabilities/configuration_ui.py @@ -290,6 +290,10 @@ def capability_configuration_editor( "select", options=explore_harness_profiles, ), + _field("composition_mode", "Composition policy", "select", options=["disabled", "explicit_only"], + description="Replan requires an exact experiment successor or typed result; it grants no execution authority."), + _field("composition_scope_id", "Research coverage scope", "text", nullable=True, + description="An opaque scope id required by the explicit-only composition policy."), ], }, "change_quality_qualification": { diff --git a/loopx/capabilities/explore/composition_frontier.py b/loopx/capabilities/explore/composition_frontier.py index cdfe6c3917..0a47b72742 100644 --- a/loopx/capabilities/explore/composition_frontier.py +++ b/loopx/capabilities/explore/composition_frontier.py @@ -213,6 +213,8 @@ def project_live_explore_composition_frontier( goal_id: str, agent_id: str | None, status_payload: Mapping[str, Any], + state_text: str | None = None, + capability_guard: Mapping[str, Any] | None = None, ) -> dict[str, Any] | None: """Read the goal's Explore graph for one live quota decision. @@ -241,6 +243,12 @@ def project_live_explore_composition_frontier( ) enabled = harness.get("enabled") is True log_path = explore_result_log_path(runtime_root, goal_id) + pinned_research = (capability_guard or {}).get("capability_id") == "explore" + if pinned_research and (not enabled or harness.get("composition_mode") != "explicit_only"): + from .research_frontier import build_research_composition_frontier, read_research_todo_history + return build_research_composition_frontier({"goal_id": goal_id, "nodes": [], "edges": []}, + candidate_sources=[], harness=harness, todos=read_research_todo_history( + runtime_root=runtime_root, goal=dict(goal), state_text=state_text), agent_id=agent_id) if not enabled: return None try: @@ -283,6 +291,15 @@ def project_live_explore_composition_frontier( if isinstance(project_asset.get("agent_todos"), Mapping) else {} ) + if harness.get("composition_mode") == "explicit_only": + from .research_frontier import build_research_composition_frontier, read_research_todo_history + return build_research_composition_frontier( + projection, candidate_sources=[ + {"node_id": event["result_id"], "research_observation": event["research_observation"]} + for event in events if event.get("research_observation") + ], harness=harness, todos=read_research_todo_history( + runtime_root=runtime_root, goal=dict(goal), state_text=state_text), agent_id=agent_id, + ) return build_explore_composition_frontier( projection, todos=_todo_items(item_todos, asset_todos), diff --git a/loopx/capabilities/explore/research_evidence.py b/loopx/capabilities/explore/research_evidence.py index a52ef6431e..59345d6de8 100644 --- a/loopx/capabilities/explore/research_evidence.py +++ b/loopx/capabilities/explore/research_evidence.py @@ -22,6 +22,16 @@ def _research_result(method: str, params: Mapping[str, Any]) -> dict[str, Any]: return result +def _research_todo_facts(todo: Mapping[str, Any]) -> dict[str, Any]: + """Bound local authority facts; never transport private task text/notes.""" + from ...control_plane.todos.todo_semantics import todo_item_is_actionable_open + fields = ("todo_id", "role", "status", "claimed_by", "replan_obligation_id", "task_class", + "action_kind", "archive_state", "excluded_agents", "explore_result_node_refs", "target_key", "capability_binding_ref", + "resume_when", "unblocks_todo_id") + return {**{field: todo[field] for field in fields if field in todo}, + "actionable_open": todo_item_is_actionable_open(dict(todo))} + + def normalize_research_observation(value: Mapping[str, Any]) -> dict[str, Any]: validate_public_safe_value(value, path="research_observation") raw = dict(value) @@ -58,6 +68,8 @@ def validate_research_append( if dict(event) != previous: raise ValueError("research observation replay cannot rewrite its node revision; use explore observe") return + if observation.get("execution_lineage"): + raise ValueError("new research execution lineage requires explore observe and a current canonical Todo read") projection = proposal or build_explore_result_projection([*events, event], goal_id=str(event["goal_id"])) _research_result("explore.research.validate_attribution", { "observation": observation, "nodes": projection["nodes"], "edges": projection["edges"], @@ -66,6 +78,7 @@ def validate_research_append( def append_research_observation( path: Path, *, goal_id: str, observation: Mapping[str, Any], agent_id: str | None = None, + registry_path: Path | None = None, runtime_root: Path | None = None, ) -> dict[str, Any]: # Use the existing append-only node revision transport. Lock attribution, # replay and append together so an input update cannot slip between them. @@ -88,6 +101,39 @@ def append_research_observation( _research_result("explore.research.validate_attribution", { "observation": canonical, "nodes": projection["nodes"], "edges": projection["edges"], }) + if canonical.get("execution_lineage"): + if registry_path is None or runtime_root is None: + raise ValueError("research execution lineage requires the selected registry and source runtime") + from ...todos import list_goal_todos + from ...history import load_registry + from ...materials import find_registry_goal + from .research_frontier import build_research_composition_frontier, read_research_todo_history + + lineage = canonical["execution_lineage"] + goal = find_registry_goal(load_registry(registry_path), goal_id) or {} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + policy = _research_result("explore.research.composition_policy", {"harness": harness}) + candidate_sources = [{"node_id": event["result_id"], "research_observation": event["research_observation"]} + for event in events if event.get("research_observation")] + frontier = None + if policy["enabled"]: + history = read_research_todo_history(runtime_root=runtime_root, goal=goal) + frontier = build_research_composition_frontier(projection, candidate_sources=candidate_sources, + harness=harness, todos=history, agent_id=agent_id) + todos = [todo for todo in history if todo["todo_id"] == lineage["successor_todo_id"]] + else: + context = list_goal_todos( + registry_path=registry_path, runtime_root_arg=str(runtime_root), goal_id=goal_id, + role="agent", todo_id=lineage["successor_todo_id"], + ) + todos = context.get("todos") or [] + todo = todos[0] if len(todos) == 1 else {} + _research_result("explore.research.validate_execution", { + "goal_id": goal_id, "agent_id": agent_id, "observation": canonical, + "nodes": projection["nodes"], "edges": projection["edges"], + "candidate_sources": candidate_sources, "frontier": frontier, + "todo": _research_todo_facts(todo), + }) event = build_explore_node_event( goal_id=goal_id, title=node["title"], node_id=node["node_id"], node_kind=node["node_kind"], status=node["status"], summary=node["summary"], blocked_reason=node["blocked_reason"], diff --git a/loopx/capabilities/explore/research_frontier.py b/loopx/capabilities/explore/research_frontier.py new file mode 100644 index 0000000000..b052946437 --- /dev/null +++ b/loopx/capabilities/explore/research_frontier.py @@ -0,0 +1,218 @@ +"""Live Explore transport. TypeScript owns policy, eligibility and lineage. + +The common replan owner retains obligation identity. No provider mutation or +work authority is performed while projecting research evidence. +""" +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from contextlib import contextmanager +from pathlib import Path +from typing import Any + +from ...control_plane.work_items.autonomous_replan_obligation import build_autonomous_replan_obligation_payload +from .research_evidence import _research_result, _research_todo_facts + + +def read_research_todo_history( + *, runtime_root: Path, goal: dict[str, Any], state_text: str | None = None, +) -> list[dict[str, Any]]: + """Read canonical active and retained rows, never a compact display list. + + The existing Todo codecs supply authority and lifecycle facts. Archives + are evidence lineage only; the typed owner decides their eligibility. + """ + from ...control_plane.coordination.local_authority import canonical_todo_items, read_canonical_todos_if_promoted + from ...control_plane.todos.active_state_todo_parser import parse_active_state_todos, parse_todo_source + from ...materials import goal_state_path + + canonical = read_canonical_todos_if_promoted(runtime_root=runtime_root, goal_id=goal["id"]) + if canonical is not None: + guards = canonical.get("goal_acceptance_work_guards") or {} + return [{**todo, **({"goal_acceptance_guard": guards[todo["todo_id"]]} if todo["todo_id"] in guards else {})} + for todo in canonical_todo_items(canonical["todos"]) if todo.get("role") != "user"] + if state_text is None: + path = goal_state_path(goal) + if path is None: + raise ValueError("research lineage requires the Goal's canonical Todo source") + state_text = path.read_text(encoding="utf-8") + active = parse_active_state_todos(state_text, item_limit=None, goal=goal).get("agent_todos", {}).get("items", []) + _, archived, _ = parse_todo_source(state_text, goal=goal) + return [*active, *[todo for todo in archived if todo.get("role") != "user"]] + + +def research_composition_obligation(gap: Mapping[str, Any], *, agent_id: str | None) -> dict[str, Any]: + guard = {"schema_version": "semantic_replan_capability_guard_v0", "capability_id": "explore", + "gap_id": gap["gap_id"], "frontier_revision": gap["frontier_revision"]} + return build_autonomous_replan_obligation_payload( + schema_version="autonomous_replan_obligation_v0", agent_id=agent_id, include_agent_id=True, + stall_threshold=0, trigger_count=1, + triggers=[{"kind": "capability_evidence_gap", "capability_id": "explore", "agent_id": agent_id, + "frontier_identity": gap["gap_id"], "frontier_revision": gap["frontier_revision"], + "text": "An explicit evidence gap requires a bound experiment and typed result."}], + guidance_actions=["create_successor", "record_typed_result"], + todo_actions=[{"action": "add", "role": "agent", "priority": "P1", + "text": gap.get("successor_summary") or "Run a bounded experiment for the explicit input set."}], + stop_condition="Respect the Goal's existing authority, budget and protected-operation gates.", + recommended_action="Bind a runnable experiment to the exact evidence gap or record its typed, attributable result.", + extra_fields={"capability_guard": guard, "satisfying_semantic_outcomes": [ + "new_runnable_successor", "capability_evidence_observed", "new_concrete_blocker", "capability_duty_retired"]}, + ) + + +def build_research_composition_frontier( + projection: Mapping[str, Any], *, candidate_sources: list[dict[str, Any]], + harness: Mapping[str, Any], todos: Sequence[Mapping[str, Any]], agent_id: str | None, +) -> dict[str, Any]: + params = {"goal_id": projection["goal_id"], "nodes": projection["nodes"], "edges": projection["edges"], + "candidate_sources": candidate_sources, "harness": dict(harness), "agent_id": agent_id} + facts = _research_result("explore.research.composition_facts", params) + bindings = [{"gap_id": gap["gap_id"], "obligation": research_composition_obligation(gap, agent_id=agent_id)} + for gap in facts["gaps"]] + frontier = _research_result("explore.research.composition", {**params, "bindings": bindings, + "todos": [_research_todo_facts(todo) for todo in todos]}) + by_gap = {binding["gap_id"]: binding["obligation"] for binding in bindings} + # These internal candidates are needed for the original Turn's settlement + # after its exact successor changes the actionable frontier. They are not + # added to the public quota packet. + transitions = [] + for gap in frontier["lineage_gaps"]: + obligation = by_gap[gap["gap_id"]] + if gap["status"] == "scheduled": + delta = {"schema_version": "replan_semantic_delta_v0", "accepted": True, + "obligation_id": obligation["obligation_id"], "outcomes": ["new_runnable_successor"], + "satisfying_outcomes": ["new_runnable_successor"], "successor_todo_id": gap["successor_todo_id"], + "capability_guard": obligation["capability_guard"], + "capability_outcome": "new_runnable_composition_experiment"} + transitions.append({"obligation": obligation, "ack": { + "schema_version": "autonomous_replan_ack_v0", "recorded": True, + "source": "todo_replan_successor_transition", "semantic_delta": delta}}) + frontier["settlement_transitions"] = transitions + selected = frontier.get("selected_gap") + frontier["obligation"] = by_gap[selected["gap_id"]] if selected else None + # Compact guard facts include every candidate, so invalidation and omitted + # display cards cannot be mistaken for a missing settled obligation. + frontier["guard_facts"] = [{"capability_guard": binding["obligation"]["capability_guard"], + "obligation_id": binding["obligation"]["obligation_id"]} for binding in bindings] + frontier["public_projection"] = public_research_composition_frontier(frontier) + return frontier + + +def public_research_composition_frontier(frontier: Mapping[str, Any] | None) -> dict[str, Any] | None: + if frontier is None: + return None + public = {key: value for key, value in frontier.items() + if key not in {"lineage_gaps", "settlement_transitions", "guard_facts", "obligation", "obligation_tasks", "public_projection"}} + if frontier.get("schema_version") == "research_composition_frontier_v0": + public["gaps"] = [{key: value for key, value in gap.items() if key != "execution_results"} + for gap in frontier["gaps"]] + if frontier.get("selected_gap"): + public["selected_gap"] = {key: value for key, value in frontier["selected_gap"].items() if key != "execution_results"} + return public + + +def prepare_research_replan_evidence( + *, runtime_root: Path, goal_id: str, agent_id: str, registry_goal: dict[str, Any] | None, + state_text: str, capability_guard: Mapping[str, Any] | None = None, +): + """IO composition root adapter; the shared gate never imports Explore.""" + import shlex + from ...control_plane.work_items.semantic_replan_writeback import CapabilityReplanEvidence + from .composition_frontier import project_live_explore_composition_frontier + + goal = {**(registry_goal or {}), "id": goal_id} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + if not (harness.get("enabled") is True and harness.get("composition_mode") == "explicit_only") and capability_guard is None: + return None + frontier = project_live_explore_composition_frontier(runtime_root=runtime_root, goal_id=goal_id, + agent_id=agent_id, state_text=state_text, capability_guard=capability_guard, + status_payload={"run_history": {"goals": [goal]}}) or {} + + def qualify(guard: Mapping[str, Any], obligation_id: str, progress: dict[str, Any] | None, runs: list[dict[str, Any]]): + if guard.get("capability_id") != "explore": + raise ValueError("selected capability writeback owner is not available") + obligation = research_composition_obligation( + {"gap_id": guard["gap_id"], "frontier_revision": guard["frontier_revision"]}, agent_id=agent_id) + if obligation["obligation_id"] != obligation_id: + raise ValueError("selected capability guard does not match its obligation identity") + delta = _research_result("explore.research.composition_writeback", {"frontier": frontier, + "capability_guard": dict(guard), "obligation_id": obligation_id, "progress_observation": progress, + "claimed_progress_fingerprints": [(run.get("progress_observation") or {}).get("fingerprint") for run in runs], + "claimed_blocker_ids": [(run.get("progress_observation") or {}).get("blocker_id") for run in runs]}) + delta["readback_actions"] = [f"loopx --runtime-root {shlex.quote(str(runtime_root))} explore summary " + f"--goal-id {shlex.quote(goal_id)} --agent-id {shlex.quote(agent_id)} --format json"] + return obligation, delta + return CapabilityReplanEvidence(frontier=frontier, qualify=qualify) + + +def attach_research_execution_projection( + projection: dict[str, Any], *, events: list[dict[str, Any]], registry: dict[str, Any], + runtime_root: Path, agent_id: str | None, +) -> None: + """Render the same live facts through existing Explore and sink fields.""" + from ...agent_registry import registered_agent_ids_for_goal + from ...materials import find_registry_goal + from ...control_plane.todos.contract import normalize_todo_claimed_by + + goal = find_registry_goal(registry, projection["goal_id"]) or {} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + if harness.get("enabled") is not True or harness.get("composition_mode") != "explicit_only": + return + registered = registered_agent_ids_for_goal(goal) + actor = normalize_todo_claimed_by(agent_id) if agent_id else registered[0] if len(registered) == 1 else None + if actor is None or actor not in registered: + raise ValueError("Live research presentation requires --agent-id naming one registered Goal agent") + frontier = build_research_composition_frontier(projection, + candidate_sources=[{"node_id": e["result_id"], "research_observation": e["research_observation"]} + for e in events if e.get("research_observation")], + harness=harness, todos=read_research_todo_history(runtime_root=runtime_root, goal=goal), agent_id=actor) + projection["research_execution_frontier"] = public_research_composition_frontier(frontier) + by_node: dict[str, list[dict[str, Any]]] = {} + for gap in frontier["lineage_gaps"]: + for node_id in [*gap["input_node_ids"], *gap["experiment_node_ids"]]: + by_node.setdefault(node_id, []).append(gap) + for node in projection["nodes"]: + gaps = by_node.get(node["node_id"], []) + if not gaps: + continue + facts = [f"Execution composition ({actor}): {gap['gap_id']} {gap['status']}" + for gap in gaps[:3]] + if len(gaps) > 3: + facts.append(f"{len(gaps) - 3} additional execution candidates omitted") + node["research_summary"] = "\n".join([node.get("research_summary") or "", *facts]).strip() + + +@contextmanager +def hold_research_completion_evidence( + *, registry_path: Path, runtime_root: Path, goal_id: str, todo: Mapping[str, Any], + state_text: str, actor_agent_id: str | None, +): + """Legacy writer transport; the same typed rule guards canonical commits.""" + from ...history import load_registry + from ...materials import find_registry_goal + from ...file_lock import exclusive_file_lock + from .result_log import build_explore_result_projection, explore_result_log_path, load_explore_result_events_strict + if not (todo.get("action_kind") == "joint_probe" or todo.get("explore_result_node_refs") + or todo.get("replan_obligation_id") or todo.get("capability_binding_ref")): + yield None + return + goal = find_registry_goal(load_registry(registry_path), goal_id) or {} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + policy = _research_result("explore.research.composition_policy", {"harness": harness}) + if not policy["enabled"] or todo.get("status") == "done" or todo.get("role") == "user": + yield None + return + todos = read_research_todo_history(runtime_root=runtime_root, goal=goal, state_text=state_text) + path = explore_result_log_path(runtime_root, goal_id) + with exclusive_file_lock(path, operation="research-legacy-completion"): + events = load_explore_result_events_strict(path, goal_id=goal_id) + projection = build_explore_result_projection(events, goal_id=goal_id) + frontier = build_research_composition_frontier(projection, + candidate_sources=[{"node_id": e["result_id"], "research_observation": e["research_observation"]} + for e in events if e.get("research_observation")], + harness=harness, todos=todos, agent_id=actor_agent_id or todo.get("claimed_by")) + guard = _research_result("explore.research.completion", {"frontier": frontier, + "todo": _research_todo_facts(todo), "actor_agent_id": actor_agent_id, "goal_id": goal_id}) + if guard["allowed"] is not True: + raise ValueError(str(guard["reason"])) + yield guard["evidence"] diff --git a/loopx/capabilities/explore/research_snapshot_host.py b/loopx/capabilities/explore/research_snapshot_host.py new file mode 100644 index 0000000000..113b9c6ac3 --- /dev/null +++ b/loopx/capabilities/explore/research_snapshot_host.py @@ -0,0 +1,45 @@ +"""Locked canonical graph transport for native completion. EOF releases locks. + +The existing Python Explore codec owns graph IO; TypeScript remains the owner +of eligibility, lineage and completion. No caller commands or task effects run. +""" +from __future__ import annotations + +import json +from pathlib import Path +import sys + +from ...file_lock import exclusive_file_lock +from .research_frontier import build_research_composition_frontier +from .result_log import build_explore_result_projection, explore_result_log_path, load_explore_result_events_strict + + +def main() -> None: + request = json.loads(sys.stdin.readline()) + root, goal = Path(request["runtime_root"]), request["goal_id"] + if not root.is_absolute(): + raise ValueError("absolute source runtime required") + path = explore_result_log_path(root, goal) + with exclusive_file_lock(path, operation="research-native-completion"): + events = load_explore_result_events_strict(path, goal_id=goal) + projection = build_explore_result_projection(events, goal_id=goal) + frontier = build_research_composition_frontier( + projection, candidate_sources=[ + {"node_id": event["result_id"], "research_observation": event["research_observation"]} + for event in events if event.get("research_observation") + ], harness=request["harness"], todos=request["todos"], agent_id=request["agent_id"], + ) + print(json.dumps({"schema_version": "research_graph_snapshot_v0", "frontier": frontier}, + ensure_ascii=False, allow_nan=False), flush=True) + # Keep the exact graph locked until the native transaction returns. + # Parent death closes this pipe and releases the OS lock automatically. + if sys.stdin.readline() != "release\n": + return + + +if __name__ == "__main__": + try: + main() + except (ValueError, OSError) as exc: + print(json.dumps({"schema_version": "research_graph_snapshot_v0", "error": type(exc).__name__}), flush=True) + raise SystemExit(1) from None diff --git a/loopx/chat_goal_configuration_api.py b/loopx/chat_goal_configuration_api.py index 5f3f2c6f79..9e76958ce0 100644 --- a/loopx/chat_goal_configuration_api.py +++ b/loopx/chat_goal_configuration_api.py @@ -112,12 +112,18 @@ def _peer_task_coordination_options(config: Mapping[str, Any]) -> dict[str, Any] def _explore_harness_options(config: Mapping[str, Any]) -> dict[str, Any]: profile = str(config.get("profile") or "").strip() or None + mode = config.get("composition_mode", "disabled") + scope = config.get("composition_scope_id") + if not isinstance(mode, str) or scope is not None and not isinstance(scope, str): + raise TypeError("Explore composition mode and scope must be strings") return { "explore_harness_enabled": _boolean_configuration( "explore_harness", config, "enabled" ), "explore_harness_profile": profile, "clear_explore_harness_profile": profile is None, + "explore_composition_mode": mode or "disabled", + "explore_composition_scope_id": scope or "", } @@ -212,7 +218,7 @@ def _goal_capability_options( }, "peer_task_coordination": {"coordinator_agent_id"}, "explore_graph": {"enabled"}, - "explore_harness": {"enabled", "profile"}, + "explore_harness": {"enabled", "profile", "composition_mode", "composition_scope_id"}, "pull_request_review": {"wait_for_ci", "review_priority"}, "change_quality_qualification": {"enabled", "safe_fix", "strict_receipt"}, "progress_review": {"mode", "signal", "drift_threshold", "contract_revision"}, diff --git a/loopx/cli_commands/explore.py b/loopx/cli_commands/explore.py index 8dfc604ab9..133cae8239 100644 --- a/loopx/cli_commands/explore.py +++ b/loopx/cli_commands/explore.py @@ -2,8 +2,9 @@ import argparse import json +from functools import partial from pathlib import Path -from typing import Callable +from typing import Any, Callable from ..capabilities.explore.result_log import ( DEFAULT_FINDING_LIMIT, @@ -133,6 +134,7 @@ def register_explore_commands( add_subcommand_format(summary) summary.add_argument("--goal-id", required=True) _add_projection_limit_args(summary) + summary.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") observe = sub.add_parser("observe", help="Record typed research evidence on an existing Explore node; grants no execution or closure authority.") add_subcommand_format(observe) @@ -147,6 +149,7 @@ def register_explore_commands( add_subcommand_format(presentation) presentation.add_argument("--goal-id", required=True) _add_projection_limit_args(presentation) + presentation.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") source_reconcile = sub.add_parser( "source-history-reconcile", @@ -172,6 +175,7 @@ def register_explore_commands( add_subcommand_format(graph) graph.add_argument("--goal-id", required=True) _add_projection_limit_args(graph) + graph.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") graph.add_argument( "--graph-format", choices=["mermaid", "json"], @@ -244,6 +248,8 @@ def _projection_for( *, runtime_root: Path, finding_limit_override: int | None = None, + registry: dict[str, Any] | None = None, + source_registry: Path | None = None, ) -> dict[str, object]: log_path = explore_result_log_path(runtime_root, args.goal_id) events = load_explore_result_events(log_path, goal_id=args.goal_id) @@ -261,6 +267,12 @@ def _projection_for( mermaid_node_limit=max(1, int(args.mermaid_node_limit)), ) projection["log_path"] = str(log_path) + if source_registry is not None: + registry = load_registry(source_registry) + if registry is not None: + from ..capabilities.explore.research_frontier import attach_research_execution_projection + attach_research_execution_projection(projection, events=events, registry=registry, + runtime_root=runtime_root, agent_id=getattr(args, "agent_id", None)) return projection @@ -328,6 +340,14 @@ def render_explore_markdown(payload: dict[str, object]) -> str: lines.append(f"- {key}: `{payload.get(key)}`") counts = payload.get("counts") research = payload.get("research_frontier") + live_research = payload.get("research_execution_frontier") + if isinstance(live_research, dict): + lines.append(f"- research execution ({live_research['agent_id']}): {live_research['pending_count']} pending, " + f"{live_research['scheduled_count']} scheduled, {live_research['observed_count']} observed, " + f"{live_research['ineligible_count']} ineligible, {live_research['dismissed_count']} dismissed, " + f"{live_research['deferred_count']} deferred") + for gap in live_research["gaps"]: + lines.append(f" - {gap['gap_id']}: {gap['status']}") if isinstance(research, dict): lines.append(f"- research composition (read-only): {research['pending_count']} pending, " f"{research['observed_count']} observed, {research['ineligible_count']} ineligible") @@ -483,6 +503,10 @@ def handle_explore_command( registry=registry, ) runtime_root = Path(str(source_runtime_route["source_runtime_root"])) + source_registry = Path(str(source_runtime_route["source_registry"])) if source_runtime_route else registry_path + same_source = source_registry.resolve() == registry_path.resolve() + projection_for = partial(_projection_for, registry=registry if same_source else None, + source_registry=None if same_source else source_registry) config_path = ( Path(args.config_path).expanduser() if getattr(args, "config_path", None) @@ -546,11 +570,13 @@ def handle_explore_command( payload = append_research_observation( explore_result_log_path(runtime_root, args.goal_id), goal_id=args.goal_id, observation=observation, agent_id=args.agent_id, + registry_path=Path(str(source_runtime_route["source_registry"])) if source_runtime_route else registry_path, + runtime_root=runtime_root, ) elif args.explore_command == "summary": - payload = _projection_for(args, runtime_root=runtime_root) + payload = projection_for(args, runtime_root=runtime_root) elif args.explore_command == "presentation": - projection = _projection_for( + projection = projection_for( args, runtime_root=runtime_root, finding_limit_override=-1, @@ -573,7 +599,7 @@ def handle_explore_command( execute=bool(args.execute), ) elif args.explore_command == "graph": - projection = _projection_for(args, runtime_root=runtime_root) + projection = projection_for(args, runtime_root=runtime_root) graph_view = build_explore_graph_view( projection.get("nodes") or [], projection.get("edges") or [], @@ -637,7 +663,7 @@ def handle_explore_command( resource_usage=resource_usage, ) else: - projection = _projection_for(args, runtime_root=runtime_root) + projection = projection_for(args, runtime_root=runtime_root) todo_payload = list_goal_todos( registry_path=registry_path, goal_id=args.goal_id, @@ -688,7 +714,7 @@ def handle_explore_command( resource_usage=resource_usage, ) else: - projection = _projection_for(args, runtime_root=runtime_root) + projection = projection_for(args, runtime_root=runtime_root) todo_payload = list_goal_todos( registry_path=registry_path, goal_id=args.goal_id, @@ -730,7 +756,7 @@ def handle_explore_command( args, config_path=config_path, runtime_root=runtime_root, - projection_for=_projection_for, + projection_for=projection_for, ) else: raise ValueError(f"unknown explore command: {args.explore_command}") diff --git a/loopx/cli_commands/explore_feishu_commands.py b/loopx/cli_commands/explore_feishu_commands.py index c8e5b26208..a45acbbef5 100644 --- a/loopx/cli_commands/explore_feishu_commands.py +++ b/loopx/cli_commands/explore_feishu_commands.py @@ -141,6 +141,7 @@ def register_explore_feishu_commands( add_subcommand_format(sync) add_config_path_arg(sync) sync.add_argument("--goal-id", required=True) + sync.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") add_projection_limit_args(sync) sync.add_argument("--base-token") for table_key in EXPLORE_TABLE_KEYS: @@ -166,6 +167,7 @@ def register_explore_feishu_commands( add_subcommand_format(card) add_config_path_arg(card) card.add_argument("--goal-id", required=True) + card.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") add_projection_limit_args(card) card.add_argument("--title") card.add_argument("--template", default="blue") diff --git a/loopx/cli_commands/project_lifecycle_inputs.py b/loopx/cli_commands/project_lifecycle_inputs.py index fd13b1413f..d7ea5aa317 100644 --- a/loopx/cli_commands/project_lifecycle_inputs.py +++ b/loopx/cli_commands/project_lifecycle_inputs.py @@ -63,6 +63,7 @@ def inline_progress_observation( args: argparse.Namespace, ) -> dict[str, object] | None: fields = { + "work_item_id": getattr(args, "progress_work_item_id", None), "surface_id": getattr(args, "progress_surface_id", None), "hypothesis_id": getattr(args, "progress_hypothesis_id", None), "probe_kind": getattr(args, "progress_probe_kind", None), diff --git a/loopx/cli_commands/project_lifecycle_refresh_state.py b/loopx/cli_commands/project_lifecycle_refresh_state.py index b316dc034e..1dbfd7bfbc 100644 --- a/loopx/cli_commands/project_lifecycle_refresh_state.py +++ b/loopx/cli_commands/project_lifecycle_refresh_state.py @@ -195,6 +195,7 @@ def register_refresh_state_command( ), ) refresh_state_parser.add_argument("--progress-surface-id") + refresh_state_parser.add_argument("--progress-work-item-id", help="Explicit typed observation source; settlement keeps its own Todo or obligation binding.") refresh_state_parser.add_argument("--progress-hypothesis-id") refresh_state_parser.add_argument("--progress-probe-kind") refresh_state_parser.add_argument("--progress-blocker-id") diff --git a/loopx/cli_commands/registry_admin.py b/loopx/cli_commands/registry_admin.py index 5a293273ec..4fc2af0864 100644 --- a/loopx/cli_commands/registry_admin.py +++ b/loopx/cli_commands/registry_admin.py @@ -484,6 +484,8 @@ def handle_registry_admin_command( explore_harness_enabled=args.explore_harness_enabled, explore_harness_profile=args.explore_harness_profile, clear_explore_harness_profile=bool(args.clear_explore_harness_profile), + explore_composition_mode=args.explore_composition_mode, + explore_composition_scope_id=args.explore_composition_scope_id, explore_graph_enabled=args.explore_graph_enabled, lark_kanban_heartbeat_sync=args.lark_kanban_heartbeat_sync, registered_agents=args.registered_agents, diff --git a/loopx/cli_commands/registry_admin_configure.py b/loopx/cli_commands/registry_admin_configure.py index 454dc62f12..4f55ed18f5 100644 --- a/loopx/cli_commands/registry_admin_configure.py +++ b/loopx/cli_commands/registry_admin_configure.py @@ -242,6 +242,14 @@ def register_configure_goal_command(subparsers: argparse._SubParsersAction) -> N action="store_true", help="Remove the goal-pinned explore profile while preserving the opt-in bit.", ) + configure_goal_parser.add_argument( + "--explore-composition-mode", choices=["disabled", "explicit_only"], + help="Opt in to typed composition obligations for this Goal; disabled preserves existing planning behavior.", + ) + configure_goal_parser.add_argument( + "--explore-composition-scope-id", + help="Opaque research coverage scope required by explicit_only composition; grants no execution authority.", + ) configure_goal_parser.add_argument( "--registered-agent", dest="registered_agents", diff --git a/loopx/cli_commands/status.py b/loopx/cli_commands/status.py index 89d9d6e72f..daa6e4167d 100644 --- a/loopx/cli_commands/status.py +++ b/loopx/cli_commands/status.py @@ -18,6 +18,7 @@ compact_agent_lane_status_payload_for_display, ) from ..control_plane.todos.contract import normalize_todo_claimed_by +from ..control_plane.quota.goal_boundary import registry_goal_by_id from ..control_plane.todos.quota_summary import ( compact_agent_lane_todos_for_status_display, ) @@ -564,7 +565,9 @@ def _sync_agent_replan_obligation_from_guard( """Keep agent-scoped status guidance identical to the quota decision.""" targets = (item, project_asset) - if not any( + obligation = guard.get("autonomous_replan_obligation") + capability_obligation = isinstance(obligation, dict) and bool(obligation.get("capability_guard")) + if not capability_obligation and not any( isinstance(target, dict) and ( "autonomous_replan_obligation" in target @@ -576,7 +579,6 @@ def _sync_agent_replan_obligation_from_guard( for target in targets ): return - obligation = guard.get("autonomous_replan_obligation") for target in targets: if not isinstance(target, dict): continue @@ -665,22 +667,38 @@ def attach_agent_lane_next_actions( "agent_member", "agent_interaction_summary", "agent_reward_memory", + "bounded_research_frontier", ), 0, ) current_agent_next_action: dict[str, Any] | None = None + goals = registry_goal_by_id(payload) for item in items: if not isinstance(item, dict): continue goal_id = str(item.get("goal_id") or "").strip() if not goal_id: continue + decision_payload = payload + frontier = None + goal = goals.get(goal_id) or {} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + if harness.get("enabled") is True and harness.get("composition_mode") == "explicit_only": + from ..capabilities.explore.composition_frontier import project_live_explore_composition_frontier + frontier = project_live_explore_composition_frontier( + runtime_root=Path(str(payload["runtime_root"])), goal_id=goal_id, + agent_id=safe_agent_id, status_payload=payload) + decision_payload = {**payload, "bounded_research_frontier": frontier} try: guard = build_quota_should_run( - payload, + decision_payload, goal_id=goal_id, agent_id=safe_agent_id, ) + selected = guard.get("replan_action_packet") or {} + obligation = (frontier or {}).get("obligation") or {} + if selected.get("capability_guard") and selected.get("obligation_id") == obligation.get("obligation_id"): + guard["autonomous_replan_obligation"] = obligation except Exception: continue next_action = guard.get("agent_lane_next_action") @@ -715,6 +733,7 @@ def attach_agent_lane_next_actions( "agent_member": agent_member, "agent_interaction_summary": interaction_summary, "agent_reward_memory": reward_memory_projection, + "bounded_research_frontier": guard.get("bounded_research_frontier"), }, ) for field in attached_fields: diff --git a/loopx/cli_commands/todo.py b/loopx/cli_commands/todo.py index af217661f2..a70f56aa91 100644 --- a/loopx/cli_commands/todo.py +++ b/loopx/cli_commands/todo.py @@ -2,7 +2,7 @@ import argparse import shlex -from collections.abc import Callable, Sequence +from collections.abc import Callable, Mapping, Sequence from operator import itemgetter from pathlib import Path @@ -86,6 +86,12 @@ ] +def _completion_hook_state_version(payload: Mapping[str, object], committed_at: str) -> str: + """Keep ordinary and terminal Todo writes distinct within one clock second.""" + continuation = payload.get("completion_continuation") + return f"{committed_at}|{continuation}" if isinstance(continuation, str) else committed_at + + def _read_todo_turn_settlement( args: argparse.Namespace, *, runtime_root: Path, ) -> QuotaSettlementReadback: @@ -209,7 +215,11 @@ def _validated_replan_successor_obligation( ), None, ) + from ..capabilities.explore.research_frontier import prepare_research_replan_evidence + capability_evidence = prepare_research_replan_evidence(runtime_root=runtime_root, goal_id=args.goal_id, + agent_id=args.claimed_by, registry_goal=registry_goal, state_text=state_text) obligation, _ = qualify_replan_writeback( + capability_evidence=capability_evidence, todo_fields=todo_fields, newest_first_runs=newest_first_runs, state_text=state_text, @@ -228,6 +238,17 @@ def _validated_replan_successor_obligation( "--replan-obligation-id does not match the current open obligation: " f"expected {current}" ) + if (obligation or {}).get("capability_guard"): + from ..capabilities.explore.research_evidence import _research_result + if capability_evidence is None: + raise ValueError("selected capability successor owner is not available") + _research_result("explore.research.composition_successor", { + "frontier": capability_evidence.frontier, "agent_id": args.claimed_by, "obligation_id": requested, + "todo": {"replan_obligation_id": requested, "claimed_by": args.claimed_by, + "action_kind": args.action_kind, "task_class": args.task_class, + "explore_result_node_refs": args.explore_result_node_refs, + "target_key": args.monitor_target_key, "status": args.status or "open", "resume_when": args.resume_when}, + }) return current @@ -751,7 +772,7 @@ def handle_todo_command( goal_id=args.goal_id, event_kind="todo_complete", identity=identity, - state_version=committed_at, + state_version=_completion_hook_state_version(payload, committed_at), committed_at=committed_at, hooks=post_writeback_hooks, projection_builder=post_writeback_projection_builder, diff --git a/loopx/cli_commands/turn.py b/loopx/cli_commands/turn.py index 8e9b908103..9ff627a3f5 100644 --- a/loopx/cli_commands/turn.py +++ b/loopx/cli_commands/turn.py @@ -1056,6 +1056,7 @@ def on_managed_start_admitted() -> None: settlement_identity, semantic_replan_guard_scoped=replan_guard_scoped, semantic_replan_obligation_id=replan_obligation_id, + semantic_replan_capability_guard=(stable_envelope.get("replan_action_packet") or {}).get("capability_guard"), ) managed_cadence = managed_cadence_start( diff --git a/loopx/configuration_catalog.py b/loopx/configuration_catalog.py index d4d9d1f4d2..f339bb7aee 100644 --- a/loopx/configuration_catalog.py +++ b/loopx/configuration_catalog.py @@ -514,13 +514,15 @@ def build_goal_configuration_catalog( "current": { "enabled": harness.get("enabled") is True, "profile": harness.get("profile"), + "composition_mode": harness.get("composition_mode", "disabled"), + "composition_scope_id": harness.get("composition_scope_id"), }, "profiles": list(explore_harness_profiles), "consider_when": ( "The goal benefits from comparing alternative branches with explicit " "evaluation criteria and guardrails." ), - "effect": "Enables read-only Explore branch and worker-lane planning.", + "effect": "Enables read-only Explore planning. Explicit composition replans require an exact experiment successor or typed result; task completion remains separate.", "does_not": [ "enable Explore Graph", "launch workers, claim todos, acquire leases, mutate state, or spend quota", diff --git a/loopx/configure_goal.py b/loopx/configure_goal.py index 248448d9ee..03686c4fb3 100644 --- a/loopx/configure_goal.py +++ b/loopx/configure_goal.py @@ -467,6 +467,8 @@ def configure_goal( explore_harness_enabled: bool | None = None, explore_harness_profile: str | None = None, clear_explore_harness_profile: bool = False, + explore_composition_mode: str | None = None, + explore_composition_scope_id: str | None = None, explore_graph_enabled: bool | None = None, registered_agents: list[str] | None = None, clear_registered_agents: bool = False, @@ -978,6 +980,8 @@ def configure_goal( or explore_harness_enabled is not None or explore_harness_profile is not None or clear_explore_harness_profile + or explore_composition_mode is not None + or explore_composition_scope_id is not None ): spawn_policy = ( goal.get("spawn_policy") @@ -1003,6 +1007,8 @@ def configure_goal( explore_harness_enabled is not None or explore_harness_profile is not None or clear_explore_harness_profile + or explore_composition_mode is not None + or explore_composition_scope_id is not None ): explore_harness = ( spawn_policy.get("explore_harness") @@ -1015,6 +1021,15 @@ def configure_goal( explore_harness.pop("profile", None) elif explore_harness_profile is not None: explore_harness["profile"] = explore_harness_profile + if explore_composition_mode is not None: + explore_harness["composition_mode"] = explore_composition_mode + if explore_composition_scope_id is not None: + if explore_composition_scope_id: + explore_harness["composition_scope_id"] = explore_composition_scope_id + else: + explore_harness.pop("composition_scope_id", None) + from .orchestration import compact_explore_harness_policy + compact_explore_harness_policy(explore_harness) if explore_harness: spawn_policy["explore_harness"] = explore_harness else: diff --git a/loopx/control_plane/capabilities/explore_research.ts b/loopx/control_plane/capabilities/explore_research.ts index 40722c1aec..1114ddd393 100644 --- a/loopx/control_plane/capabilities/explore_research.ts +++ b/loopx/control_plane/capabilities/explore_research.ts @@ -45,10 +45,11 @@ function digest(value: unknown): string { .sort(([a], [b]) => compare(a, b)).map(([key, child]) => [key, stable(child)])) : item; return createHash("sha256").update(JSON.stringify(stable(value))).digest("hex").slice(0, 16); } +export {id as researchIdentifier, digest as researchDigest}; export function normalizeResearchObservation(params: JsonObject): JsonObject { const raw = object(params.observation, "research observation", - ["schema_version", "explore_node_id", "progress", "closure_basis", "composition_candidates", "input_observations", "fingerprint"]); + ["schema_version", "explore_node_id", "progress", "closure_basis", "composition_candidates", "input_observations", "execution_lineage", "composition_resolution", "fingerprint"]); requireStringLiteral(raw.schema_version, [OBSERVATION], "research observation schema"); const node = id(raw.explore_node_id, "explore_node_id"); // The Python transport composes its existing generic progress codec. This @@ -109,8 +110,47 @@ export function normalizeResearchObservation(params: JsonObject): JsonObject { if (new Set(lineage.map(input => input.node_id)).size !== lineage.length || lineage.some(input => input.node_id === node)) { throw new EffectRuntimeRequestError("input observations must have distinct non-self node identities"); } + let execution: JsonObject | null = null; + if (raw.execution_lineage !== undefined) { + const value = object(raw.execution_lineage, "execution_lineage", ["schema_version", "goal_id", "gap_id", "replan_obligation_id", "successor_todo_id", "agent_id"]); + requireStringLiteral(value.schema_version, ["research_execution_lineage_v0"], "execution lineage schema"); + execution = {schema_version: value.schema_version}; + for (const field of ["goal_id", "gap_id", "replan_obligation_id", "successor_todo_id", "agent_id"]) { + execution[field] = id(value[field], `execution_lineage.${field}`); + } + if (!/^research-composition-[a-f0-9]{16}$/.test(String(execution.gap_id)) + || !/^replan-[a-f0-9]{16}$/.test(String(execution.replan_obligation_id)) + || !/^todo_[A-Za-z0-9_-]+$/.test(String(execution.successor_todo_id)) + || progress.work_item_id !== execution.successor_todo_id || lineage.length !== 2) { + throw new EffectRuntimeRequestError("execution lineage requires exact gap, obligation, Todo and binary input identities"); + } + } + let resolution: JsonObject | null = null; + if (raw.composition_resolution !== undefined) { + const value = object(raw.composition_resolution, "composition_resolution", + ["schema_version", "disposition", "basis", "evidence_ids"]); + requireStringLiteral(value.schema_version, ["research_composition_resolution_v0"], "composition resolution schema"); + const disposition = requireStringLiteral(value.disposition, ["dismissed", "deferred"], "resolution disposition"); + const refs = ids(value.evidence_ids, "resolution evidence_ids"); + if (execution === null || !refs.length || refs.some(ref => !evidence.includes(ref))) { + throw new EffectRuntimeRequestError("composition resolution requires execution lineage and attributable evidence"); + } + resolution = {schema_version: value.schema_version, disposition, evidence_ids: refs}; + if (disposition === "dismissed") { + if (result !== "no_followup" || closure?.disposition !== "no_followup") { + throw new EffectRuntimeRequestError("candidate dismissal requires coverage-backed no_followup and its closure basis"); + } + resolution.basis = requireStringLiteral(value.basis, ["duplicate", "invalid", "unsafe", "outside_scope"], "dismissal basis"); + } else if (result !== "blocked" || value.basis !== undefined + || progress.coverage_complete === true || closure?.disposition === "no_followup" || closure?.disposition === "exhausted" + || !/^todo_[A-Za-z0-9_-]+$/.test(String(progress.blocker_id))) { + throw new EffectRuntimeRequestError("temporary composition deferral requires blocked progress with a canonical blocker Todo id"); + } + } const canonical: JsonObject = {schema_version: OBSERVATION, explore_node_id: node, progress, ...(lineage.length ? {input_observations: lineage} : {}), + ...(execution ? {execution_lineage: execution} : {}), + ...(resolution ? {composition_resolution: resolution} : {}), ...(closure ? {closure_basis: closure} : {}), composition_candidates: candidates}; return {...canonical, fingerprint: digest(canonical)}; } @@ -124,7 +164,7 @@ function observationFor(node: JsonObject): JsonObject | null { function eligible(node: JsonObject | undefined): boolean { if (!node || !["resolved", "dead_end"].includes(String(node.status))) return false; const observation = observationFor(node); - return !!observation && TERMINAL.has(String((observation.progress as JsonObject).result_class)); + return !!observation && !observation.composition_resolution && TERMINAL.has(String((observation.progress as JsonObject).result_class)); } function evidenceFor(node: JsonObject): Set { const observation = observationFor(node); @@ -164,7 +204,8 @@ export function validateResearchAttribution(params: JsonObject): JsonObject { return observation; } -export function projectResearchFrontier(params: JsonObject): JsonObject { +/** Full internal candidate set; presentation compaction cannot hide a write gate. */ +export function researchCompositionGaps(params: JsonObject): JsonObject[] { const goal = id(params.goal_id, "goal_id"); const nodeRows = rows(params.nodes, "nodes"); const nodes = new Map(nodeRows.map(node => [String(node.node_id), node])); @@ -206,23 +247,35 @@ export function projectResearchFrontier(params: JsonObject): JsonObject { return sets.every(set => refs.some(ref => set.has(ref))) && refs.every(ref => sets.some(set => set.has(ref))); }); const experiments = experimentsByInputs.get(JSON.stringify(candidate.inputs)) ?? []; - const observed = allEligible && attributable && experiments.some(experiment => { - if (!eligible(experiment)) return false; - const lineage = observationFor(experiment)?.input_observations as JsonObject[] ?? []; + const terminal = allEligible && attributable ? experiments.filter(experiment => { + const outcome = observationFor(experiment); + if (!["resolved", "dead_end"].includes(String(experiment.status)) + || !outcome || !TERMINAL.has(String((outcome.progress as JsonObject).result_class))) return false; + const lineage = outcome.input_observations as JsonObject[] ?? []; return JSON.stringify(lineage.map(input => input.node_id)) === JSON.stringify(candidate.inputs) && lineage.every(input => observationFor(nodes.get(String(input.node_id))!)?.fingerprint === input.fingerprint); - }); + }) : []; + const observed = terminal.some(experiment => !observationFor(experiment)?.composition_resolution); + const dismissed = terminal.some(experiment => experiment.status === "dead_end" + && (observationFor(experiment)?.composition_resolution as JsonObject)?.disposition === "dismissed"); const active = experiments.filter(node => ["open", "exploring"].includes(String(node.status))); gaps.push({gap_id: `research-composition-${identity}`, input_node_ids: candidate.inputs, input_observations: inputs.filter((node): node is JsonObject => !!node).map(node => ({node_id: node.node_id, fingerprint: observationFor(node)?.fingerprint ?? null})), - state: !allEligible || !attributable ? "ineligible" : observed ? "observed" : "pending", + state: !allEligible || !attributable ? "ineligible" : observed ? "observed" : dismissed ? "dismissed" : "pending", reason: !allEligible ? "terminal_input_observation_required" : !attributable ? "input_evidence_invalidated" - : observed ? "typed_experiment_outcome" : "joint_experiment_result_required", + : observed ? "typed_experiment_outcome" : dismissed ? "typed_candidate_dismissal" : "joint_experiment_result_required", interaction_kinds: [...new Set(candidate.claims.map(claim => String(claim.interaction_kind)))].sort(), experiment_node_ids: experiments.map(node => String(node.node_id)).sort(), active_experiment_node_ids: active.map(node => String(node.node_id)).sort()}); } + return gaps; +} + +export function projectResearchFrontier(params: JsonObject): JsonObject { + const goal = id(params.goal_id, "goal_id"); + const nodeRows = rows(params.nodes, "nodes"); + const gaps = researchCompositionGaps(params); const count = (state: string) => gaps.filter(gap => gap.state === state).length; const gapsByInput = new Map(); for (const gap of gaps) for (const input of gap.input_node_ids as string[]) { @@ -239,12 +292,14 @@ export function projectResearchFrontier(params: JsonObject): JsonObject { progress ? `Research: ${String(progress.result_class).replaceAll("_", " ")}` : "Research: observation invalidated", ...(progress?.coverage_scope_id ? [`coverage ${progress.coverage_scope_id}`] : []), ...(basis ? [`closure ${basis.disposition}`] : []), + ...(observation?.composition_resolution ? [`composition resolution ${(observation.composition_resolution as JsonObject).disposition}`] : []), ...(related.length ? [`composition ${related.filter(gap => gap.state === "pending").length} pending, ${related.filter(gap => gap.state === "ineligible").length} ineligible, ${related.filter(gap => gap.state === "observed").length} observed`] : []), ].join("; ")}; }); return {schema_version: "research_frontier_projection_v0", goal_id: goal, mode: "read_only_shadow", candidate_count: gaps.length, pending_count: count("pending"), observed_count: count("observed"), ineligible_count: count("ineligible"), projected_count: Math.min(gaps.length, MAX_RESEARCH_GAPS), + ...(nodeRows.some(node => observationFor(node)?.composition_resolution) ? {dismissed_count: count("dismissed")} : {}), omitted_count: Math.max(0, gaps.length - MAX_RESEARCH_GAPS), gaps: gaps.slice(0, MAX_RESEARCH_GAPS), grants_execution_authority: false, node_summaries: nodeSummaries}; } diff --git a/loopx/control_plane/capabilities/explore_research_execution.ts b/loopx/control_plane/capabilities/explore_research_execution.ts new file mode 100644 index 0000000000..93fa859087 --- /dev/null +++ b/loopx/control_plane/capabilities/explore_research_execution.ts @@ -0,0 +1,334 @@ +/** Explore execution attribution. This records evidence, never work authority. */ +import type {JsonObject} from "../effect_program.ts"; +import {EffectRuntimeRequestError} from "../effect_runtime_errors.ts"; +import {requireJsonObject} from "../runtime_decode.ts"; +import {normalizeResearchObservation, researchCompositionGaps, researchIdentifier, researchDigest, + validateResearchAttribution} from "./explore_research.ts"; +import {evaluateTodoResumeConditions, TODO_RESUME_EVALUATION_REQUEST_SCHEMA_VERSION, + resumeConditionHasKnownPendingTarget} from "../todos/resume_condition.ts"; + +const object = (value: unknown): JsonObject => value && typeof value === "object" && !Array.isArray(value) ? value as JsonObject : {}; +const array = (value: unknown): JsonObject[] => Array.isArray(value) ? value.map(row => requireJsonObject(row, "research row")) : []; +const same = (left: unknown, right: unknown) => JSON.stringify(left) === JSON.stringify(right); +type CompositionFrontierState = "disabled" | "pending" | "scheduled" | "observed" | "dismissed" | "deferred" | "ineligible" | "empty"; + +export function normalizeResearchCompositionPolicy(params: JsonObject): JsonObject { + const harness = requireJsonObject(params.harness, "Explore harness"); + const mode = harness.composition_mode ?? "disabled"; + if (!["disabled", "explicit_only"].includes(String(mode))) { + throw new EffectRuntimeRequestError("composition_mode must be disabled or explicit_only"); + } + const scope = harness.composition_scope_id == null || harness.composition_scope_id === "" ? null + : researchIdentifier(harness.composition_scope_id, "composition_scope_id"); + if (mode === "explicit_only" && scope === null) { + throw new EffectRuntimeRequestError("explicit_only composition requires composition_scope_id"); + } + return {schema_version: "research_composition_policy_v0", mode, coverage_scope_id: scope, + enabled: harness.enabled === true && mode === "explicit_only"}; +} + +function boundTodo(todo: JsonObject, gap: JsonObject, obligationId: unknown, agent: unknown, + retainedHistory = false): boolean { + const refs = todo.explore_result_node_refs; + const completed = retainedHistory && todo.status === "done"; + return (todo.claimed_by === agent || completed && todo.claimed_by == null) && todo.replan_obligation_id === obligationId + && (!todo.role || todo.role === "agent") + && todo.task_class === "advancement_task" && todo.action_kind === "joint_probe" + && (completed || todo.archive_state !== "archive") && Array.isArray(refs) && refs.length === 1 + && (gap.experiment_node_ids as string[]).includes(String(refs[0])) + && (!todo.target_key || todo.target_key === refs[0]) + && !(Array.isArray(todo.excluded_agents) && todo.excluded_agents.includes(agent)); +} + +/** Internal unbounded facts for obligation identity transport. Never an extra + * public list: CLI/status receive only the compact frontier below. */ +export function researchCompositionFacts(params: JsonObject): JsonObject { + const policy = normalizeResearchCompositionPolicy(params); + if (!policy.enabled) return {policy, gaps: []}; + const gaps = researchCompositionGaps(params).map(gap => ({...gap, + frontier_revision: `research-composition-v0:${researchDigest([policy, gap.gap_id, gap.input_observations])}`})); + return {policy, gaps}; +} + +/** Join current evidence and canonical Todo facts. Only exact lineage can + * schedule or observe a gap; historical diagnostic observations stay cold. */ +export function projectResearchComposition(params: JsonObject): JsonObject { + const {policy, gaps: raw} = researchCompositionFacts(params); + const gaps = raw as JsonObject[], todos = array(params.todos), bindings = array(params.bindings); + const nodes = array(params.nodes); + const nodeById = new Map(nodes.map(node => [node.node_id, node])); + const todoById = new Map(todos.map(todo => [todo.todo_id, todo])); + if (todoById.size !== todos.length || nodeById.size !== nodes.length) { + throw new EffectRuntimeRequestError("canonical research node and Todo identities must be unique"); + } + const todosByObligation = new Map(); + for (const todo of todos) { + // A terminal record may clear its claim. Its accepted observation retains + // the actor; this historical join cannot grant runnable work authority. + if (todo.claimed_by !== params.agent_id && !(todo.status === "done" && todo.claimed_by == null)) continue; + const owned = todosByObligation.get(todo.replan_obligation_id) ?? []; + owned.push(todo); todosByObligation.set(todo.replan_obligation_id, owned); + } + const bindingByGap = new Map(bindings.map(binding => [binding.gap_id, requireJsonObject(binding.obligation, "composition obligation")])); + const resumeByTodo = new Map(array((policy as JsonObject).enabled ? evaluateTodoResumeConditions({ + schema_version: TODO_RESUME_EVALUATION_REQUEST_SCHEMA_VERSION, + items: todos, source_items: todos, kinds: ["todo_done"], rollout_events: [], + }).conditions : []).map(row => [row.todo_id, object(row.condition)])); + const projected: JsonObject[] = []; + for (const gap of gaps) { + const obligation = bindingByGap.get(gap.gap_id); + if (!obligation || !/^replan-[a-f0-9]{16}$/.test(String(obligation.obligation_id))) { + throw new EffectRuntimeRequestError("composition gap requires its common obligation identity"); + } + const matched = (todosByObligation.get(obligation.obligation_id) ?? []) + .filter(todo => boundTodo(todo, gap, obligation.obligation_id, params.agent_id, true)); + const results = (gap.experiment_node_ids as string[]).map(id => nodeById.get(id)) + .flatMap(node => { + if (!node?.research_observation || gap.state === "ineligible") return []; + const observation = normalizeResearchObservation({observation: node.research_observation}); + const lineage = object(observation.execution_lineage), progress = object(observation.progress); + const todo = todoById.get(lineage.successor_todo_id); + const resolution = object(observation.composition_resolution); + const attributable = !!todo && matched.includes(todo) + && (["open", "claimed", "done"].includes(String(todo.status)) + || resolution.disposition === "deferred" && ["blocked", "deferred"].includes(String(todo.status))) + && lineage.goal_id === params.goal_id && lineage.agent_id === params.agent_id && lineage.gap_id === gap.gap_id + && lineage.replan_obligation_id === obligation.obligation_id + && progress.work_item_id === todo.todo_id && progress.coverage_scope_id === (policy as JsonObject).coverage_scope_id + && same(observation.input_observations, gap.input_observations); + return attributable ? [{node, observation, progress, todo, resolution}] : []; + }); + const observed = results.find(result => ["resolved", "dead_end"].includes(String(result.node.status)) + && !result.resolution.disposition && ["exploration_exhausted", "no_followup"].includes(String(result.progress.result_class))); + const dismissed = results.find(result => result.node.status === "dead_end" && result.resolution.disposition === "dismissed"); + const deferred = results.find(result => { + if (result.node.status !== "blocked" || result.resolution.disposition !== "deferred" + || !["blocked", "deferred"].includes(String(result.todo.status))) return false; + const blocker = todoById.get(result.progress.blocker_id), condition = resumeByTodo.get(result.todo.todo_id); + return !!blocker && blocker.role === "agent" && blocker.task_class === "blocker" + && blocker.claimed_by === params.agent_id && blocker.archive_state !== "archive" + && ["open", "deferred"].includes(String(blocker.status)) && blocker.unblocks_todo_id === result.todo.todo_id + && !!condition && condition.kind === "todo_done" && condition.target_todo_id === blocker.todo_id + && resumeConditionHasKnownPendingTarget(condition, result.todo); + }); + const successor = gap.state !== "ineligible" ? matched.find(todo => todo.actionable_open === true + && boundTodo(todo, gap, obligation.obligation_id, params.agent_id) && ["open", "claimed"].includes(String(todo.status)) + && ["open", "exploring"].includes(String(nodeById.get((todo.explore_result_node_refs as string[])[0])?.status))) : undefined; + const experiment = successor ? (successor.explore_result_node_refs as string[])[0] + : (gap.active_experiment_node_ids as string[])[0] ?? null; + projected.push({...gap, status: gap.state === "ineligible" ? "ineligible" : observed ? "observed" : dismissed ? "dismissed" + : successor ? "scheduled" : deferred ? "deferred" : "pending", + obligation_id: obligation.obligation_id, experiment_node_ref: experiment, + input_node_refs: gap.input_node_ids, required_outcome: "typed_joint_experiment_result", + successor_todo_id: successor?.todo_id ?? null, + observed_todo_id: observed?.todo?.todo_id ?? null, + observed_progress_fingerprint: observed?.progress.fingerprint ?? null, + dismissed_todo_id: dismissed?.todo.todo_id ?? null, + deferred_todo_id: deferred?.todo.todo_id ?? null, + blocker_todo_id: deferred?.progress.blocker_id ?? null, + resume_when: deferred?.todo.resume_when ?? null, + execution_results: results.map(result => ({todo_id: result.todo?.todo_id, experiment_node_id: result.node.node_id, + observation_fingerprint: result.observation.fingerprint, progress_fingerprint: result.progress.fingerprint, + progress: result.progress, + result_class: result.progress.result_class, blocker_id: result.progress.blocker_id ?? null, + disposition: result.resolution.disposition ?? null, + evidence_ids: result.progress.evidence_ids})), + successor_summary: experiment ? `Run the bounded joint experiment: ${experiment}` : "Create one binary experiment for this explicit input set, then bind a runnable Todo.", + successor_binding: experiment ? {action_kind: "joint_probe", task_domain: "research", target_key: experiment, + explore_result_node_refs: [experiment]} : null}); + } + projected.sort((a, b) => Number(a.status !== "pending") - Number(b.status !== "pending") + || String(a.gap_id).localeCompare(String(b.gap_id), "en")); + const selected = projected.find(gap => gap.status === "pending") ?? null; + const state: CompositionFrontierState = !(policy as JsonObject).enabled ? "disabled" : selected ? "pending" + : projected.some(gap => gap.status === "scheduled") ? "scheduled" + : projected.some(gap => gap.status === "deferred") ? "deferred" + : projected.some(gap => gap.status === "ineligible") ? "ineligible" + : projected.some(gap => gap.status === "observed") ? "observed" : projected.length ? "dismissed" : "empty"; + return {schema_version: "research_composition_frontier_v0", goal_id: params.goal_id, agent_id: params.agent_id, policy, + grants_execution_authority: false, + enabled: (policy as JsonObject).enabled, state, + candidate_count: gaps.length, pending_count: projected.filter(gap => gap.status === "pending").length, + scheduled_count: projected.filter(gap => gap.status === "scheduled").length, + observed_count: projected.filter(gap => gap.status === "observed").length, + dismissed_count: projected.filter(gap => gap.status === "dismissed").length, + deferred_count: projected.filter(gap => gap.status === "deferred").length, + ineligible_count: projected.filter(gap => gap.status === "ineligible").length, + omitted_count: Math.max(0, projected.length - 3), gaps: projected.slice(0, 3), selected_gap: selected, + // This internal lane is removed by the transport before public projection. + lineage_gaps: projected, + obligation_tasks: todos.filter(todo => todo.replan_obligation_id) + .map(todo => ({obligation_id: todo.replan_obligation_id, todo_id: todo.todo_id, + status: todo.status, archive_state: todo.archive_state ?? "active"}))}; +} + +function retirementContract(frontier: JsonObject, guard: JsonObject, obligation: unknown): JsonObject | null { + if (frontier.schema_version !== "research_composition_frontier_v0" + || guard.schema_version !== "semantic_replan_capability_guard_v0" || guard.capability_id !== "explore") return null; + const current = array(frontier.lineage_gaps).find(row => row.gap_id === guard.gap_id); + if (frontier.enabled === true && (!current || current.frontier_revision === guard.frontier_revision + && current.status !== "ineligible")) return null; + const blocking = array(frontier.obligation_tasks).filter(todo => todo.obligation_id === obligation + && todo.archive_state !== "archive" && ["open", "claimed"].includes(String(todo.status))); + const revision = `capability-evidence-v0:${researchDigest([frontier.policy, current ?? null])}`; + const blocker = `capability-invalidated-${researchDigest([guard, revision])}`; + return {schema_version: "capability_obligation_retirement_v0", disposition: "invalidated", + reason_code: frontier.enabled !== true ? "source_disabled" : current?.status === "ineligible" + ? "source_ineligible" : "source_revision_changed", + capability_id: "explore", obligation_id: obligation, original_guard: guard, current_revision: revision, + blocking_todo_ids: blocking.slice(0, 3).map(todo => todo.todo_id), blocking_todo_count: blocking.length, + progress_observation: {schema_version: "typed_progress_observation_v0", result_class: "blocked", + work_item_id: obligation, blocker_id: blocker, evidence_ids: [revision]}}; +} + +/** Qualify the exact research duty selected at admission. Generic progress or + * a vision/read/ACK cannot impersonate the canonical experiment observation. */ +export function qualifyResearchCompositionWriteback(params: JsonObject): JsonObject { + const frontier = requireJsonObject(params.frontier, "live research frontier"); + const guard = requireJsonObject(params.capability_guard, "selected capability guard"); + const gap = array(frontier.lineage_gaps).find(row => row.gap_id === guard.gap_id + && row.frontier_revision === guard.frontier_revision && row.obligation_id === params.obligation_id); + const observation = object(params.progress_observation); + const progressFingerprint = observation.fingerprint; + const repeated = Array.isArray(params.claimed_progress_fingerprints) && params.claimed_progress_fingerprints.includes(progressFingerprint); + let outcome: string | null = null, capabilityOutcome: string | null = null; + let result: JsonObject | undefined; + const retirement = gap && gap.status !== "ineligible" ? null : retirementContract(frontier, guard, params.obligation_id); + let retired = false; + if (retirement && retirement.blocking_todo_count === 0 && typeof progressFingerprint === "string") { + const claimed = {...observation}; delete claimed.fingerprint; + if (researchDigest(claimed) === researchDigest(retirement.progress_observation)) { + outcome = "capability_duty_retired"; capabilityOutcome = "composition_duty_invalidated"; retired = true; + } + } + if (guard.capability_id === "explore" && frontier.enabled === true && gap) { + if (gap.status === "scheduled" && gap.successor_todo_id) { + outcome = "new_runnable_successor"; capabilityOutcome = "new_runnable_composition_experiment"; + } else if (!repeated && typeof progressFingerprint === "string") { + result = array(gap.execution_results).find(row => row.progress_fingerprint === progressFingerprint + && row.todo_id === observation.work_item_id && researchDigest(row.progress) === researchDigest(observation)); + if (result && gap.status === "observed" && result.todo_id === gap.observed_todo_id + && progressFingerprint === gap.observed_progress_fingerprint) { + outcome = "capability_evidence_observed"; capabilityOutcome = "composition_experiment_observed"; + } else if (result && gap.status === "dismissed" && result.todo_id === gap.dismissed_todo_id + && result.disposition === "dismissed") { + outcome = "capability_evidence_observed"; capabilityOutcome = "composition_candidate_dismissed"; + } else if (result && gap.status === "deferred" && result.todo_id === gap.deferred_todo_id + && result.disposition === "deferred" && result.blocker_id === gap.blocker_todo_id + && !(Array.isArray(params.claimed_blocker_ids) && params.claimed_blocker_ids.includes(result.blocker_id))) { + outcome = "new_concrete_blocker"; capabilityOutcome = "composition_temporarily_deferred"; + } + } + } + return {schema_version: "replan_semantic_delta_v0", obligation_id: params.obligation_id, + accepted: outcome !== null, outcomes: outcome ? [outcome] : [], satisfying_outcomes: outcome ? [outcome] : [], + required_any_of: ["new_runnable_successor", "capability_evidence_observed", "new_concrete_blocker", "capability_duty_retired"], + capability_guard: guard, capability_outcome: capabilityOutcome, + ...(retirement ? {retirement_contract: retirement} : {}), + ...(retired ? {retirement: {...retirement, progress_fingerprint: progressFingerprint}} : {}), + ...(gap?.successor_todo_id ? {successor_todo_id: gap.successor_todo_id} : {}), + observation_fingerprint: result?.observation_fingerprint ?? null, + reason_code: outcome ? "research_semantic_delta_accepted" : !gap || frontier.enabled !== true + ? "research_frontier_invalidated" : repeated ? "research_result_replayed" : "research_result_required", + reason: outcome ? "the selected evidence duty has an exact canonical transition" + : "the selected evidence duty requires a current bound experiment result or runnable successor; reads, ACKs, stale inputs and unrelated progress cannot settle it"}; +} + +export function validateResearchCompositionSuccessor(params: JsonObject): JsonObject { + const frontier = requireJsonObject(params.frontier, "live research frontier"), todo = requireJsonObject(params.todo, "successor intent"); + const gap = object(frontier.selected_gap); + const refs = todo.explore_result_node_refs; + const accepted = frontier.enabled === true && gap.status === "pending" + && todo.replan_obligation_id === gap.obligation_id && boundTodo(todo, gap, gap.obligation_id, params.agent_id) + && ["open", "claimed"].includes(String(todo.status)) && !todo.resume_when + && Array.isArray(refs) && (gap.active_experiment_node_ids as string[]).includes(String(refs[0])); + if (!accepted) throw new EffectRuntimeRequestError( + `research successor ${gap.obligation_id ?? params.obligation_id}: bind one current binary experiment with joint_probe and no deferral; read Explore summary before todo add`); + return {accepted: true, gap_id: gap.gap_id, obligation_id: gap.obligation_id}; +} + +/** A completion guard is eligibility evidence, never a replacement for actor, + * lease or CAS admission. Historical terminal replays do not execute new work. */ +export function qualifyResearchCompletion(params: JsonObject): JsonObject { + const frontier = requireJsonObject(params.frontier, "research completion frontier"); + const todo = requireJsonObject(params.todo, "completion Todo"); + const gaps = array(frontier.lineage_gaps); + const required = frontier.enabled === true && todo.status !== "done" && todo.role !== "user" + && (todo.action_kind === "joint_probe" || todo.capability_binding_ref === "explore:research-composition-v0" + || gaps.some(gap => gap.obligation_id === todo.replan_obligation_id)); + if (!required) return {required: false, allowed: true, evidence: null}; + const actor = params.actor_agent_id ?? todo.claimed_by; + for (const gap of gaps) { + if (!["observed", "dismissed"].includes(String(gap.status)) || !boundTodo(todo, gap, gap.obligation_id, actor)) continue; + const result = array(gap.execution_results).find(row => row.todo_id === todo.todo_id + && ["exploration_exhausted", "no_followup"].includes(String(row.result_class)) + && (todo.explore_result_node_refs as string[]).includes(String(row.experiment_node_id))); + if (!result) continue; + return {required: true, allowed: true, evidence: { + schema_version: "capability_completion_evidence_v0", capability_id: "explore", goal_id: params.goal_id, + todo_id: todo.todo_id, agent_id: actor, gap_id: gap.gap_id, obligation_id: gap.obligation_id, + frontier_revision: gap.frontier_revision, experiment_node_id: result.experiment_node_id, + input_observations: gap.input_observations, observation_fingerprint: result.observation_fingerprint, + progress_fingerprint: result.progress_fingerprint, evidence_ids: result.evidence_ids, + disposition: result.disposition === "dismissed" ? "candidate_dismissed" : "experiment_observed", + }}; + } + return {required: true, allowed: false, reason_code: "research_experiment_result_required", + reason: "This Todo requires its current typed experiment observation with exact task/input lineage before closeout. Read Explore summary and record the result through explore observe.", + evidence: null}; +} + +/** The transport supplies a fresh canonical Todo snapshot and its existing + * actionable-open decision. Historical replay bypasses this gate without + * rewriting evidence; any later settlement must revalidate its own authority. + */ +export function validateResearchExecution(params: JsonObject): JsonObject { + const observation = validateResearchAttribution(params); + const lineage = requireJsonObject(observation.execution_lineage, "execution_lineage"); + const todo = requireJsonObject(params.todo, "canonical execution Todo"); + const node = (params.nodes as JsonObject[]).find(row => row.node_id === observation.explore_node_id); + const reject = (reason: string): never => { + throw new EffectRuntimeRequestError(`research execution ${lineage.replan_obligation_id}: ${reason}; read the current Todo and Explore summary before explore observe`); + }; + const frontier = params.frontier == null ? null : requireJsonObject(params.frontier, "execution frontier"); + if (frontier && (frontier.schema_version !== "research_composition_frontier_v0" + || frontier.goal_id !== params.goal_id || frontier.agent_id !== params.agent_id)) { + reject("the current execution frontier must belong to this Goal and actor"); + } + // M2 diagnostics retain their cold meaning. M3 uses the same current + // canonical lineage join as status and closeout, including retained Todos. + const live = frontier?.enabled === true; + const gap = (live ? array(frontier!.lineage_gaps) : researchCompositionGaps(params)) + .find(row => row.gap_id === lineage.gap_id); + const uncovered = live ? !!gap && ["pending", "scheduled"].includes(String(gap.status)) + && gap.obligation_id === lineage.replan_obligation_id + && object(frontier!.policy).coverage_scope_id === object(observation.progress).coverage_scope_id + : gap?.state === "pending"; + if (lineage.goal_id !== params.goal_id || lineage.agent_id !== params.agent_id) { + reject("Goal or actor differs from execution lineage"); + } + if (todo.todo_id !== lineage.successor_todo_id || todo.claimed_by !== lineage.agent_id + || todo.replan_obligation_id !== lineage.replan_obligation_id + || todo.task_class !== "advancement_task" || todo.action_kind !== "joint_probe" + || (todo.role && todo.role !== "agent") + || todo.actionable_open !== true || todo.archive_state === "archive" + || (Array.isArray(todo.excluded_agents) && todo.excluded_agents.includes(lineage.agent_id))) { + reject("a current same-agent runnable joint-probe Todo with the exact obligation is required"); + } + const refs = todo.explore_result_node_refs; + if (!Array.isArray(refs) || refs.length !== 1 || refs[0] !== observation.explore_node_id + || (todo.target_key && todo.target_key !== observation.explore_node_id)) { + reject("Todo must bind exactly this experiment node"); + } + if (node?.node_kind !== "experiment" || (node.agent_id && node.agent_id !== lineage.agent_id) + || !gap || !uncovered + || !(gap.experiment_node_ids as string[]).includes(String(observation.explore_node_id)) + || JSON.stringify(gap.input_observations) !== JSON.stringify(observation.input_observations)) { + reject("the experiment must cover this pending gap's current exact input observations"); + } + const progress = observation.progress as JsonObject; + if (progress.result_class === "unchanged" || !(progress.evidence_ids as string[]).length) { + reject("an execution result needs typed progress and evidence; a read or ACK is insufficient"); + } + return observation; +} diff --git a/loopx/control_plane/capabilities/explore_research_terminal.ts b/loopx/control_plane/capabilities/explore_research_terminal.ts new file mode 100644 index 0000000000..af08bab1bb --- /dev/null +++ b/loopx/control_plane/capabilities/explore_research_terminal.ts @@ -0,0 +1,93 @@ +/** Native completion's fixed graph IO adapter. The model supplies no process, + * snapshot, authority or approval fields; the provider supplies the actual Todo. */ +import {spawn, type ChildProcessWithoutNullStreams} from "node:child_process"; +import {readFile} from "node:fs/promises"; +import {createHash} from "node:crypto"; +import {isAbsolute} from "node:path"; +import {fileURLToPath} from "node:url"; +import {createInterface} from "node:readline"; +import type {JsonObject} from "../effect_program.ts"; +import type {TodoTerminalEvidenceGuard} from "../coordination/todo_terminal_lifecycle.ts"; +import {requireJsonObject} from "../runtime_decode.ts"; +import {normalizeResearchCompositionPolicy, qualifyResearchCompletion} from "./explore_research_execution.ts"; + +export class ResearchTerminalEvidenceHost { + private child: ChildProcessWithoutNullStreams | null = null; + private exited: Promise | null = null; + private readonly root: string; + private readonly goal: string; + private readonly source: JsonObject; + constructor(root: string, goal: string, source: JsonObject) {this.root = root; this.goal = goal; this.source = source;} + + readonly qualify: TodoTerminalEvidenceGuard = async context => { + if (context.todo.role === "user" || context.todo.status === "done" + || context.todo.action_kind !== "joint_probe" + && context.todo.capability_binding_ref !== "explore:research-composition-v0" + && !(Array.isArray(context.todo.explore_result_node_refs) && context.todo.explore_result_node_refs.length) + && !context.todo.replan_obligation_id) return {required: false, allowed: true, evidence: null}; + const source = requireJsonObject(this.source, "research registry source"); + const bytes = await readFile(String(source.path)); + if (createHash("sha256").update(bytes).digest("hex") !== source.sha256) { + return {allowed: false, reason_code: "authority_source_changed", reason: "Goal configuration changed before completion; read the current source."}; + } + const registry = requireJsonObject(JSON.parse(bytes.toString("utf8")), "research registry"); + const goals = Array.isArray(registry.goals) ? registry.goals : []; + const matches = goals.filter((value): value is JsonObject => value !== null && typeof value === "object" + && !Array.isArray(value)).filter(goal => goal.id === this.goal); + if (matches.length > 1) throw new Error("ambiguous research Goal source"); + const spawnPolicy = matches[0]?.spawn_policy as JsonObject | undefined; + const harness = spawnPolicy?.explore_harness as JsonObject | undefined; + if (!harness || !normalizeResearchCompositionPolicy({harness}).enabled) return {required: false, allowed: true, evidence: null}; + const python = process.env.LOOPX_EFFECT_RUNTIME_PYTHON; + if (!python || !isAbsolute(python) || python.includes("\0")) { + return {allowed: false, reason_code: "research_evidence_host_required", + reason: "Native research completion needs the source-selected Python adapter context; use the installed LoopX entrypoint."}; + } + this.child = spawn(python, ["-m", "loopx.capabilities.explore.research_snapshot_host"], { + stdio: ["pipe", "pipe", "pipe"], env: {...process.env, PYTHONPATH: fileURLToPath(new URL("../../../", import.meta.url))}, + }); + this.exited = new Promise(resolve => this.child!.once("close", () => resolve())); + this.child.on("error", () => {}); this.child.stdin.on("error", () => {}); this.child.stderr.resume(); + const replies = createInterface({input: this.child.stdout})[Symbol.asyncIterator](); + this.child.stdin.write(JSON.stringify({runtime_root: this.root, goal_id: this.goal, + harness, agent_id: context.actor_agent_id ?? context.todo.claimed_by, + todos: context.todos.map(todo => Object.fromEntries([ + "todo_id", "role", "status", "claimed_by", "replan_obligation_id", "task_class", "action_kind", + "archive_state", "excluded_agents", "explore_result_node_refs", "target_key", + "resume_when", "unblocks_todo_id", + ].filter(field => Object.hasOwn(todo, field)).map(field => [field, todo[field]]))), + }) + "\n"); + let timer: ReturnType | undefined; + try { + const line = await Promise.race([replies.next().then(row => row.done ? null : row.value), + this.exited.then(() => null), new Promise(resolve => {timer = setTimeout(() => resolve(null), 20000);})]); + if (line === null || Buffer.byteLength(line, "utf8") > 2 * 1024 * 1024) { + return {allowed: false, reason_code: "research_graph_snapshot_unavailable", reason: "Current locked research graph is unavailable; retry its source read before closeout."}; + } + const snapshot = requireJsonObject(JSON.parse(line), "research graph snapshot"); + if (snapshot.schema_version !== "research_graph_snapshot_v0" || snapshot.error) { + if (snapshot.error === "LockAcquireTimeoutError") return {allowed: false, reason_code: "research_graph_busy", + reason: "An Explore writer owns the graph lock; retry this same completion identity after the writer finishes."}; + return {allowed: false, reason_code: "research_graph_snapshot_invalid", reason: "The current research graph is invalid; repair its canonical source before closeout."}; + } + return qualifyResearchCompletion({frontier: snapshot.frontier, todo: context.todo, + actor_agent_id: context.actor_agent_id, goal_id: context.goal_id}); + } finally {clearTimeout(timer);} + }; + + current(): boolean { + // A child that exited no longer owns the kernel lock. Check this again at + // the transaction's existing source-admission boundaries, including CAS. + return this.child === null || this.child.exitCode === null && this.child.signalCode === null; + } + + async close(): Promise { + if (this.child) { + this.child.stdin.end("release\n"); + // The host exits after releasing its kernel lock. A bounded forced exit + // also closes the descriptors if its process cannot finish normally. + const timer = setTimeout(() => this.child?.kill(), 1000); + try {await this.exited;} finally {clearTimeout(timer);} + } + } +} diff --git a/loopx/control_plane/coordination/local_authority_runtime.ts b/loopx/control_plane/coordination/local_authority_runtime.ts index 66b36e1d67..c6872cafbe 100644 --- a/loopx/control_plane/coordination/local_authority_runtime.ts +++ b/loopx/control_plane/coordination/local_authority_runtime.ts @@ -2,6 +2,7 @@ import {requirePromotionRegisteredAgents} from "./shadow_registry_source.ts"; import {readPromotionReceipt, commitPromotionAndReadBack} from './promotion_receipt.ts'; import {reviewedPromotionPlan, promotionPlanDigest, decodeReviewedPromotionOperation, REVIEWED_PROMOTION_OPERATION_RESULT_SCHEMA} from './reviewed_promotion_plan.ts'; import {registryAuthoritySourceCheck} from "./authority_source.ts"; +import {ResearchTerminalEvidenceHost} from "../capabilities/explore_research_terminal.ts"; import {decodeTaskLeaseProof} from "./task_lease_proof.ts"; import {COORDINATION_TODO_ARCHIVE_RESULT_SCHEMA} from "./todo_archive.ts"; import {readCoordinationOwnership} from "./ownership_observation.ts"; @@ -1299,6 +1300,8 @@ export async function terminalLifecycleLocalCoordinationTodo( const store = await openRuntimeStore(root, goalId, dependencies); sourceAuthority = sourceAuthorityFor(store); providerEvidence.source_authority = sourceAuthority; + const researchEvidence = new ResearchTerminalEvidenceHost(root, goalId, requireJsonObject(input.registry_source, "registry source")); + try { return {...await executeCoordinationTodoTerminalLifecycle(store, { validation_source_provider_revision: input.validation_source_provider_revision == null ? null : requireAuthorityStoreId(input.validation_source_provider_revision, "validation source provider revision"), @@ -1357,7 +1360,8 @@ export async function terminalLifecycleLocalCoordinationTodo( ? null : requireJsonObject(input.completion_policy_request, "completion_policy_request"), dry_run: input.dry_run as boolean, now: claimObservedAt(input.observed_at), - }, authoritySourcesCurrent), ...providerEvidence}; + }, async () => await authoritySourcesCurrent() && researchEvidence.current(), researchEvidence.qualify), ...providerEvidence}; + } finally {await researchEvidence.close();} }); } catch (error) { return {schema_version: COORDINATION_TODO_TERMINAL_LIFECYCLE_RESULT_SCHEMA, diff --git a/loopx/control_plane/coordination/todo_blocked_lifecycle.ts b/loopx/control_plane/coordination/todo_blocked_lifecycle.ts index d6610e0498..fb0f426583 100644 --- a/loopx/control_plane/coordination/todo_blocked_lifecycle.ts +++ b/loopx/control_plane/coordination/todo_blocked_lifecycle.ts @@ -1,4 +1,4 @@ -/** A user-directed pause is a Todo lifecycle change, not a completed delivery. +/** A bounded pause is a Todo lifecycle change, not a completed delivery. * Retire only an inactive execution generation in the same provider CAS; a * live holder must release its lease before another actor can pause the work. */ import type {JsonObject} from "../effect_program.ts"; @@ -7,17 +7,23 @@ import type {CoordinationTodoUpdateInput} from "./todo_update_intent.ts"; import {canonicalTaskLease} from "./task_lease_state.ts"; import {leaseEpoch, leaseIsActive, leaseVersion} from "../work_items/task_lease_acquire.ts"; import {releasedTaskLeaseRecord} from "../work_items/task_lease_lifecycle_decision.ts"; +import {normalizeTodoResumeWhen, TODO_RESUME_NORMALIZE_REQUEST_SCHEMA_VERSION} from "../todos/resume_condition.ts"; -const LIFECYCLE_FIELDS = new Set(["status", "reason", "clear_resume_when"]); +const LIFECYCLE_FIELDS = new Set(["status", "reason", "clear_resume_when", "resume_when"]); export function isBlockedLifecycleTransition( input: CoordinationTodoUpdateInput, todo: JsonObject, ): boolean { const intent = input.planning_intent ?? {}; + const pause = todo.status === "open" && intent.status === "blocked"; + const clearWait = intent.clear_resume_when === true && intent.resume_when == null; + const typedWait = pause && intent.clear_resume_when !== true && typeof intent.resume_when === "string" + && normalizeTodoResumeWhen({schema_version: TODO_RESUME_NORMALIZE_REQUEST_SCHEMA_VERSION, + resume_when: intent.resume_when}) === intent.resume_when; return todo.role === "agent" && - ((todo.status === "open" && intent.status === "blocked") || + (pause || (todo.status === "blocked" && intent.status === "open")) && - intent.clear_resume_when === true && + (clearWait || typedWait) && typeof intent.reason === "string" && Boolean(intent.reason.trim()) && Object.keys(input.patch).length === 0 && input.clear_fields.length === 0 && Object.keys(intent).every(field => LIFECYCLE_FIELDS.has(field)); diff --git a/loopx/control_plane/coordination/todo_terminal_lifecycle.ts b/loopx/control_plane/coordination/todo_terminal_lifecycle.ts index fb53fe13d3..e50d0a6bf4 100644 --- a/loopx/control_plane/coordination/todo_terminal_lifecycle.ts +++ b/loopx/control_plane/coordination/todo_terminal_lifecycle.ts @@ -1,5 +1,11 @@ import {planUserCompletion} from "../todos/user_completion.ts"; import {AUTHORITY_SOURCE_CHANGED, uncheckedAuthoritySource, type AuthoritySourceCheck} from "./authority_source.ts"; + +/** Service-owned evidence admission, evaluated against the actual provider + * head. Public requests cannot supply this callback or an approval boolean. */ +export type TodoTerminalEvidenceGuard = (context: { + goal_id: string; todo: JsonObject; todos: JsonObject[]; actor_agent_id: string | null; command: string; +}) => Promise; import {normalizeTodoUpdateInput, prepareUpdatedTodo, type CoordinationTodoUpdateInput, type TodoCompletionEdit} from "./todo_update_intent.ts"; import {todoUpdateAdmissionRejection} from "./todo_update_admission.ts"; import { createHash } from "node:crypto"; @@ -1024,6 +1030,7 @@ export async function executeCoordinationTodoTerminalLifecycle( store: AuthorityStore, rawInput: CoordinationTodoTerminalLifecycleInput, authoritySourcesCurrent: AuthoritySourceCheck = uncheckedAuthoritySource, + terminalEvidenceGuard: TodoTerminalEvidenceGuard | null = null, ): Promise { let normalized: CoordinationTodoTerminalLifecycleInput; try { @@ -1221,6 +1228,18 @@ export async function executeCoordinationTodoTerminalLifecycle( {goal_acceptance_guard: acceptance}, "decision_rejection"); } const acceptanceRequirements = acceptanceCompletionRequirements(completionHead, input.goal_id, input.todo_id); + let capabilityCompletionEvidence: JsonObject | null = null; + if (terminalEvidenceGuard !== null && !(authority.outcome === "no_change" && todo.status === "done")) { + const evidenceGuard = await terminalEvidenceGuard({goal_id: input.goal_id, todo, + todos: [...projection.todos.values()], actor_agent_id: input.actor_agent_id, command: input.command}); + if (evidenceGuard.allowed !== true) { + return terminalFailure(String(evidenceGuard.reason_code ?? "capability_completion_evidence_required"), + String(evidenceGuard.reason ?? "Current capability evidence is required before closeout"), + {capability_completion_guard: evidenceGuard}, "decision_rejection"); + } + capabilityCompletionEvidence = evidenceGuard.evidence == null ? null + : canonicalAuthorityObject(evidenceGuard.evidence, "capability completion evidence"); + } const acceptanceBinding = acceptanceRequirements === null ? null : acceptanceSourceBinding(input, acceptanceRequirements, head.provider_revision); let acceptanceEvidence: JsonObject | null = null; @@ -1591,6 +1610,7 @@ export async function executeCoordinationTodoTerminalLifecycle( completed_at: target.todo.completed_at, ...(completionResult === null ? {} : {completion_result: completionResult}), ...(acceptanceEvidence === null ? {} : {goal_acceptance_completion: acceptanceEvidence}), + ...(capabilityCompletionEvidence === null ? {} : {capability_completion_evidence: capabilityCompletionEvidence}), // A preview that omits this would show an unconditional close for work the // real call still gates. Name the criteria the real call must run; never // their argv, which stays out of every projection. diff --git a/loopx/control_plane/effect_runtime.py b/loopx/control_plane/effect_runtime.py index a815707358..7da554fe6a 100644 --- a/loopx/control_plane/effect_runtime.py +++ b/loopx/control_plane/effect_runtime.py @@ -8,6 +8,7 @@ import shutil import socket import subprocess +import sys import tempfile import time import uuid @@ -285,6 +286,10 @@ def _runtime_fingerprint_for_snapshot( snapshot: _RuntimeSourceSnapshot, ) -> str: digest = hashlib.sha256() + # The daemon's fixed Python adapter context must follow the interpreter + # selected by this caller, even when two environments share source files. + digest.update(sys.executable.encode("utf-8")) + digest.update(str(sys.version_info[:3]).encode("ascii")) source_root = Path(root) paths = (source_root / relative for relative, *_metadata in snapshot) # Reads may finish out of order; hash the same relative names and original @@ -845,6 +850,8 @@ def _start_runtime(*, fingerprint: str, info_path: Path) -> dict[str, Any]: token = secrets.token_urlsafe(32) environment = os.environ.copy() environment["LOOPX_EFFECT_RUNTIME_TOKEN"] = token + # Fixed daemon context, never a model-supplied executable argument. + environment["LOOPX_EFFECT_RUNTIME_PYTHON"] = sys.executable # Capture stderr so a rejected startup can publish a typed # configuration diagnostic instead of a bare exit status. The capture # is an unlinked temporary file, so it cannot deadlock the child on a diff --git a/loopx/control_plane/effect_runtime_handlers.ts b/loopx/control_plane/effect_runtime_handlers.ts index cf04abcb75..b3eabd6818 100644 --- a/loopx/control_plane/effect_runtime_handlers.ts +++ b/loopx/control_plane/effect_runtime_handlers.ts @@ -2,6 +2,9 @@ import {manageNewGoalStorage} from "./coordination/local_authority_defaults.ts"; import {deriveAgentOperationActor, managedOperationBindingCurrent, normalizeAgentOperationExecutor, planAgentOperationHandoff, projectAgentOperationInbox, projectManagedOperationTransport} from "./work_items/operation_agent_handoff.ts"; import {projectDecisionNotice} from "./presentation/decision_notice.ts"; import {normalizeResearchObservation, validateResearchAttribution, projectResearchFrontier} from "./capabilities/explore_research.ts"; +import {validateResearchExecution, normalizeResearchCompositionPolicy, researchCompositionFacts, + projectResearchComposition, qualifyResearchCompositionWriteback, + validateResearchCompositionSuccessor, qualifyResearchCompletion} from "./capabilities/explore_research_execution.ts"; import {projectTodoSummary} from "./todos/summary_projection.ts"; import {admitAutomationStart, confirmAutomationStart, manageAutomationCadence, projectCadenceSchedule} from "./quota/automation_cadence.ts"; import {deliverShadowEntry} from "./coordination/shadow_entry_delivery.ts"; @@ -874,6 +877,13 @@ export function createEffectRuntimeHandlers( ["work_item.replan_semantics.project", projectReplanSemantics], ["explore.research.normalize", normalizeResearchObservation], ["explore.research.validate_attribution", validateResearchAttribution], + ["explore.research.validate_execution", validateResearchExecution], + ["explore.research.composition_policy", normalizeResearchCompositionPolicy], + ["explore.research.composition_facts", researchCompositionFacts], + ["explore.research.composition", projectResearchComposition], + ["explore.research.composition_writeback", qualifyResearchCompositionWriteback], + ["explore.research.composition_successor", validateResearchCompositionSuccessor], + ["explore.research.completion", qualifyResearchCompletion], ["explore.research.frontier", projectResearchFrontier], ["work_item.replan_history.project", projectReplanHistory], ["work_item.replan_history.project_snapshot", projectReplanHistorySnapshot], diff --git a/loopx/control_plane/goals/goal_frontier/__init__.py b/loopx/control_plane/goals/goal_frontier/__init__.py index ea0b108b8e..ebfc350f53 100644 --- a/loopx/control_plane/goals/goal_frontier/__init__.py +++ b/loopx/control_plane/goals/goal_frontier/__init__.py @@ -1018,6 +1018,7 @@ def derive_goal_frontier_replan_obligation_from_summaries( current_transition_replan_ack: dict[str, Any] | None = None, acceptance_gaps: list[dict[str, Any]] | None = None, monitor_lane_semantically_valid: bool = True, + capability_obligation: dict[str, Any] | None = None, ) -> dict[str, Any] | None: """Return a compact replan obligation when the goal frontier has no advancement. @@ -1120,6 +1121,7 @@ def derive_goal_frontier_replan_obligation_from_summaries( current_agent_blocker_count=safe_non_negative_int( (agent_todo_summary or {}).get("current_agent_blocker_count") ), + capability_gap_pending=bool(capability_obligation and capability_obligation.get("required") is True), monitor_no_change_streak_triggered=( monitor_no_change_trigger is not None ), @@ -1139,6 +1141,8 @@ def derive_goal_frontier_replan_obligation_from_summaries( ) if not replan_rule.derives_obligation: return None + if replan_rule.rule is GoalFrontierReplanRule.CAPABILITY_EVIDENCE_GAP: + return capability_obligation if replan_rule.rule is GoalFrontierReplanRule.TODO_SUCCESSION_GAP: settlement_items = succession_gap_items[:3] settlement_todo_ids = [ @@ -1686,8 +1690,13 @@ def build_goal_frontier_projection_context_from_status( monitor_lane_semantically_valid=not goal_vision_state_is_closed( (latest_agent_vision or {}).get("state") ), + capability_obligation=(status_payload.get("bounded_research_frontier") or {}).get("obligation"), ) - frontier_transition_ack = replan_successor_transition_ack( + capability_transitions = (status_payload.get("bounded_research_frontier") or {}).get("settlement_transitions") or [] + frontier_transition_ack = next((candidate.get("ack") for candidate in capability_transitions + if candidate.get("obligation", {}).get("obligation_id") == (frontier_replan_obligation or {}).get("obligation_id")), None) if ( + frontier_replan_obligation or {} + ).get("capability_guard") else replan_successor_transition_ack( agent_todo_summary, agent_id=agent_id, replan_obligation=frontier_replan_obligation, @@ -1773,7 +1782,7 @@ def build_goal_frontier_projection_context_from_status( "projected_replan_ack": projected_replan_ack, "replan_transition_ack": replan_transition_ack, "run_replan_transition_ack": run_replan_transition_ack, - "replan_transition_candidates": [run_transition_candidate, frontier_transition_candidate], + "replan_transition_candidates": [run_transition_candidate, frontier_transition_candidate, *capability_transitions], } diff --git a/loopx/control_plane/goals/goal_frontier/replan_rules.py b/loopx/control_plane/goals/goal_frontier/replan_rules.py index 0b105b10c2..9ab75c8f1d 100644 --- a/loopx/control_plane/goals/goal_frontier/replan_rules.py +++ b/loopx/control_plane/goals/goal_frontier/replan_rules.py @@ -28,8 +28,11 @@ class GoalFrontierReplanRule(str, Enum): DUE_MONITOR_EXECUTION = "due_monitor_execution" FUTURE_MONITOR_WAIT = "future_monitor_wait" MONITOR_FRONTIER_EXHAUSTED = "monitor_frontier_exhausted" + CAPABILITY_EVIDENCE_GAP = "capability_evidence_gap" +# Stable presentation indices for the v0 wire contract. Optional rules append +# here; their actual precedence is the explicit table in the interpreter. GOAL_FRONTIER_REPLAN_RULE_ORDER = tuple(GoalFrontierReplanRule) @@ -51,6 +54,7 @@ class GoalFrontierReplanFacts: outcome_checkpoint_replan_required: bool = False long_todo_chain_triggered: bool = False current_agent_blocker_count: int = 0 + capability_gap_pending: bool = False monitor_no_change_streak_triggered: bool = False monitor_only_lane: bool = False monitor_count: int = 0 @@ -143,6 +147,12 @@ def select_goal_frontier_replan_rule( False, "an explicit current-agent blocker owns the empty frontier", ), + ( + GoalFrontierReplanRule.CAPABILITY_EVIDENCE_GAP, + facts.capability_gap_pending and facts.selectable_frontier_advancement == 0, + True, + "a caller-owned evidence gap remains without selectable advancement", + ), ( GoalFrontierReplanRule.MONITOR_NO_CHANGE_STREAK, facts.monitor_only_lane diff --git a/loopx/control_plane/quota/heartbeat_receipt.py b/loopx/control_plane/quota/heartbeat_receipt.py index f02fec2e14..e350d203c5 100644 --- a/loopx/control_plane/quota/heartbeat_receipt.py +++ b/loopx/control_plane/quota/heartbeat_receipt.py @@ -172,6 +172,7 @@ def ensure_turn_heartbeat_settlement_receipt( *, semantic_replan_guard_scoped: bool, semantic_replan_obligation_id: str | None, + semantic_replan_capability_guard: Mapping[str, object] | None = None, ) -> dict[str, object]: """Idempotently bind a Turn-created quota guard to its settlement identity. @@ -239,6 +240,14 @@ def ensure_turn_heartbeat_settlement_receipt( raise HeartbeatReceiptIdentityConflictError( "Turn heartbeat receipt belongs to another semantic replan guard" ) + if semantic_replan_capability_guard is not None and any(effective_details.get( + source + ) != semantic_replan_capability_guard[field] for source, field in ( + ("semantic_replan_capability_id", "capability_id"), + ("semantic_replan_gap_id", "gap_id"), + ("semantic_replan_frontier_revision", "frontier_revision"), + )): + raise HeartbeatReceiptIdentityConflictError("Turn heartbeat receipt belongs to another capability guard") return effective details = { @@ -253,6 +262,10 @@ def ensure_turn_heartbeat_settlement_receipt( details["semantic_replan_obligation_id"] = ( normalized_semantic_replan_obligation_id or "" ) + if semantic_replan_capability_guard is not None: + details["semantic_replan_capability_id"] = semantic_replan_capability_guard["capability_id"] + details["semantic_replan_gap_id"] = semantic_replan_capability_guard["gap_id"] + details["semantic_replan_frontier_revision"] = semantic_replan_capability_guard["frontier_revision"] source_event_id = ( str(effective.get("event_id") or "").strip() if effective is not None @@ -534,6 +547,13 @@ def heartbeat_receipt_view( receipt["semantic_replan_obligation_id"] = ( semantic_replan_obligation_id ) + if details.get("semantic_replan_capability_id"): + receipt["semantic_replan_capability_guard"] = { + "schema_version": "semantic_replan_capability_guard_v0", + "capability_id": details["semantic_replan_capability_id"], + "gap_id": details.get("semantic_replan_gap_id"), + "frontier_revision": details.get("semantic_replan_frontier_revision"), + } pending_action_todo_id = heartbeat_receipt_pending_action_todo_id(event) if pending_action_todo_id: receipt["pending_action_selection"] = { diff --git a/loopx/control_plane/quota/settlement.py b/loopx/control_plane/quota/settlement.py index f5317506cc..ab8757983c 100644 --- a/loopx/control_plane/quota/settlement.py +++ b/loopx/control_plane/quota/settlement.py @@ -151,7 +151,7 @@ class QuotaSettlementReadback: terminal_closeout: SettlementResult[dict[str, Any]] terminal_settlement: SettlementResult[dict[str, Any]] workspace_causality: dict[str, str] | None - semantic_replan_guard: dict[str, str | None] | None + semantic_replan_guard: dict[str, Any] | None writeback_run: dict[str, Any] | None spend_run: dict[str, Any] | None heartbeat_receipt: dict[str, Any] | None @@ -217,6 +217,8 @@ def render_settlement_progress_markdown(payload: dict[str, Any]) -> list[str]: lines = [f"- settlement: `{progress.get('state')}`"] if progress.get("closeout_kind") == "typed_blocked_writeback_no_spend": lines.append("- closeout: typed blocked writeback; no quota slot spent") + elif progress.get("closeout_kind") == "capability_duty_retired_no_spend": + lines.append("- closeout: exact invalidated capability duty retired; no quota slot spent") owed = payload.get("settlement_owed") if isinstance(owed, dict): lines.extend([f"- settlement_owed: {owed['reason']}", "", "```sh", owed["command"], "```"]) @@ -272,7 +274,7 @@ def _optional_readback_record(value: Any) -> dict[str, Any] | None: return dict(value) -def _semantic_replan_guard(value: Any) -> dict[str, str | None] | None: +def _semantic_replan_guard(value: Any) -> dict[str, Any] | None: guard = _optional_readback_record(value) if guard is None: return None @@ -292,11 +294,19 @@ def _semantic_replan_guard(value: Any) -> dict[str, str | None] | None: or legacy_guard_claims_selection ): raise RuntimeError("TypeScript semantic replan guard shape mismatch") - return { + result: dict[str, Any] = { "schema_version": SEMANTIC_REPLAN_GUARD_SCHEMA, "scope": str(scope), "selected_obligation_id": selected_obligation_id, } + if "selected_capability_guard" in guard: + capability = guard["selected_capability_guard"] + if not isinstance(capability, dict) or selected_obligation_id is None: + raise RuntimeError("selected capability guard requires a typed object and obligation") + # The TS readback owner validates the schema and ids. Preserve the + # selected authority evidence when adapting the receipt to refresh. + result["selected_capability_guard"] = dict(capability) + return result def read_heartbeat_settlement( diff --git a/loopx/control_plane/quota/settlement_cli.py b/loopx/control_plane/quota/settlement_cli.py index 2096d20ca4..a2c8b1cef0 100644 --- a/loopx/control_plane/quota/settlement_cli.py +++ b/loopx/control_plane/quota/settlement_cli.py @@ -375,6 +375,15 @@ def quota_rollout_details( "quiet_noop_allowed": bool(agent_channel.get("quiet_noop_allowed")), "closeout_required": closeout_required, } + replan_packet = payload.get("replan_action_packet") + receipt = payload.get("heartbeat_receipt") + capability_guard = (replan_packet.get("capability_guard") if isinstance(replan_packet, Mapping) else None) or ( + receipt.get("semantic_replan_capability_guard") if isinstance(receipt, Mapping) else None + ) + if isinstance(capability_guard, Mapping): + details["semantic_replan_capability_id"] = capability_guard["capability_id"] + details["semantic_replan_gap_id"] = capability_guard["gap_id"] + details["semantic_replan_frontier_revision"] = capability_guard["frontier_revision"] cli_channel = interaction.get("cli_channel") if isinstance(cli_channel, Mapping) and cli_channel.get("quota_spend_source"): details["quota_spend_source"] = cli_channel["quota_spend_source"] diff --git a/loopx/control_plane/quota/settlement_phase.ts b/loopx/control_plane/quota/settlement_phase.ts index b2ff8d2efd..9c7e75a2e2 100644 --- a/loopx/control_plane/quota/settlement_phase.ts +++ b/loopx/control_plane/quota/settlement_phase.ts @@ -17,6 +17,45 @@ export function isBoundedBlockedRetry(value: unknown, todoId: string | null): bo return Number.isFinite(delay) && delay >= 60 && delay <= 30 * 60; } +/** A service-qualified capability duty can retire its exact admitted Turn. + * The caller first verifies the durable writeback receipt. No Task/Goal + * completion, progress or debit follows from this lifecycle evidence. */ +export function isCapabilityRetirementWriteback(value: unknown, identity: SettlementIdentity, selectedGuard: unknown): boolean { + const run = jsonObject(value), guard = jsonObject(selectedGuard); + const ack = jsonObject(run?.autonomous_replan_ack), delta = jsonObject(ack?.semantic_delta); + const retirement = jsonObject(delta?.retirement), original = jsonObject(retirement?.original_guard); + const deltaGuard = jsonObject(delta?.capability_guard), progress = jsonObject(run?.progress_observation); + const sameGuard = (candidate: Record | null) => !!guard && !!candidate + && candidate.schema_version === "semantic_replan_capability_guard_v0" + && candidate.capability_id === guard.capability_id && candidate.gap_id === guard.gap_id + && candidate.frontier_revision === guard.frontier_revision; + return identity.binding_kind === "autonomous_replan" && identity.replan_obligation_id !== null + && !!run && run.goal_id === identity.goal_id && run.agent_id === identity.agent_id + && run.turn_instance_id === identity.turn_instance_id && run.replan_obligation_id === identity.replan_obligation_id + && run.delivery_outcome === "outcome_gap" && ack?.recorded === true + && delta?.schema_version === "replan_semantic_delta_v0" && delta.accepted === true + && delta.obligation_id === identity.replan_obligation_id + && JSON.stringify(delta.outcomes) === '["capability_duty_retired"]' + && JSON.stringify(delta.satisfying_outcomes) === '["capability_duty_retired"]' + && retirement?.schema_version === "capability_obligation_retirement_v0" && retirement.disposition === "invalidated" + && retirement.obligation_id === identity.replan_obligation_id && retirement.capability_id === guard?.capability_id + && ["source_disabled", "source_ineligible", "source_revision_changed"].includes(String(retirement.reason_code)) + && retirement.blocking_todo_count === 0 && Array.isArray(retirement.blocking_todo_ids) && retirement.blocking_todo_ids.length === 0 + && typeof retirement.current_revision === "string" && retirement.current_revision.length > 0 + && sameGuard(original) && sameGuard(deltaGuard) + && progress?.schema_version === "typed_progress_observation_v0" && progress.result_class === "blocked" + && progress.work_item_id === identity.replan_obligation_id + && typeof progress.blocker_id === "string" && progress.blocker_id.length > 0 + && typeof progress.fingerprint === "string" && progress.fingerprint.length > 0 + && retirement.progress_fingerprint === progress.fingerprint + && jsonObject(retirement.progress_observation)?.blocker_id === progress.blocker_id + && jsonObject(retirement.progress_observation)?.work_item_id === identity.replan_obligation_id + && jsonObject(retirement.progress_observation)?.result_class === "blocked" + && jsonObject(retirement.progress_observation)?.schema_version === "typed_progress_observation_v0" + && JSON.stringify(jsonObject(retirement.progress_observation)?.evidence_ids) === JSON.stringify(progress.evidence_ids) + && JSON.stringify(progress.evidence_ids) === JSON.stringify([retirement.current_revision]); +} + /** The committed checkpoint accepts progress for a Turn, not Todo completion. * The caller must first verify this writeback's exact durable receipt. */ export function isAcceptedInFlightWriteback( diff --git a/loopx/control_plane/quota/settlement_readback.ts b/loopx/control_plane/quota/settlement_readback.ts index b55ce95f3a..359fd02e93 100644 --- a/loopx/control_plane/quota/settlement_readback.ts +++ b/loopx/control_plane/quota/settlement_readback.ts @@ -33,6 +33,7 @@ import { isBoundedBlockedRetry, isCommittedMonitorPollEffect, isAcceptedInFlightWriteback, + isCapabilityRetirementWriteback, receiptBoundMonitorPhase, receiptBoundReplayPhase, } from "./settlement_phase.ts"; @@ -118,7 +119,7 @@ function settlementProgress( identity: SettlementResult, writeback: SettlementResult, spend: SettlementResult, writebackRun: JsonObject | null, spendRun: JsonObject | null, spendSource: unknown = "heartbeat", - blockedNoSpend = false, + noSpendKind: "typed_blocked_writeback_no_spend" | "capability_duty_retired_no_spend" | null = null, ): JsonObject { const source = spendSource ?? "heartbeat"; if (source !== "heartbeat" && source !== "visible-goal") { @@ -126,16 +127,16 @@ function settlementProgress( } const state: SettlementProgressState = identity.failure ? "identity_required" : writeback.failure ? (writebackRun ? "writeback_receipt_required" : "writeback_required") - : blockedNoSpend ? "settled" + : noSpendKind ? "settled" : spend.failure ? (spendRun ? "spend_receipt_required" : "spend_required") : "settled"; return { schema_version: "quota_settlement_progress_v0", state, next_step: identity.failure ? "validation" : writeback.failure ? "durable_writeback" - : blockedNoSpend ? null : spend.failure ? "quota_spend" : null, + : noSpendKind ? null : spend.failure ? "quota_spend" : null, quota_spend_source: source, - ...(blockedNoSpend ? { - closeout_kind: "typed_blocked_writeback_no_spend", + ...(noSpendKind ? { + closeout_kind: noSpendKind, } : {}), }; } @@ -406,7 +407,10 @@ function runEffectMatches( export function projectSemanticReplanGuard( receiptDetails: JsonObject, ): JsonObject { + const hasCapability = ["semantic_replan_capability_id", "semantic_replan_gap_id", "semantic_replan_frontier_revision"] + .some(field => Object.hasOwn(receiptDetails, field)); if (!Object.hasOwn(receiptDetails, "semantic_replan_obligation_id")) { + if (hasCapability) throw new EffectRuntimeRequestError("a capability guard requires a selected obligation", "malformed_settlement_state"); return { schema_version: SEMANTIC_REPLAN_GUARD_SCHEMA, scope: "legacy_unscoped", @@ -415,6 +419,9 @@ export function projectSemanticReplanGuard( } const rawObligationId = receiptDetails.semantic_replan_obligation_id; if (rawObligationId === "" || rawObligationId === null) { + if (hasCapability) { + throw new EffectRuntimeRequestError("a capability guard requires a selected obligation", "malformed_settlement_state"); + } return { schema_version: SEMANTIC_REPLAN_GUARD_SCHEMA, scope: "turn_guard", @@ -432,9 +439,26 @@ export function projectSemanticReplanGuard( schema_version: SEMANTIC_REPLAN_GUARD_SCHEMA, scope: "turn_guard", selected_obligation_id: selectedObligationId, + ...(hasCapability ? { + selected_capability_guard: decodeCapabilityGuard({schema_version: "semantic_replan_capability_guard_v0", + capability_id: receiptDetails.semantic_replan_capability_id, gap_id: receiptDetails.semantic_replan_gap_id, + frontier_revision: receiptDetails.semantic_replan_frontier_revision}), + } : {}), }; } +function decodeCapabilityGuard(value: unknown): JsonObject { + const guard = jsonObject(value); + if (!guard || guard.schema_version !== "semantic_replan_capability_guard_v0" + || typeof guard.capability_id !== "string" || !/^[a-z][a-z0-9-]{0,63}$/.test(guard.capability_id) + || typeof guard.gap_id !== "string" || !/^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/.test(guard.gap_id) + || typeof guard.frontier_revision !== "string" || !/^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/.test(guard.frontier_revision) + || Object.keys(guard).some(key => !["schema_version", "capability_id", "gap_id", "frontier_revision"].includes(key))) { + throw new EffectRuntimeRequestError("heartbeat receipt capability guard is malformed", "malformed_settlement_state"); + } + return guard; +} + function runMatchesBinding(run: JsonObject, identity: SettlementIdentity): boolean { return normalizeTodoId(run.todo_id) === identity.todo_id && normalizeReplanObligationId(run.replan_obligation_id) === @@ -1008,6 +1032,9 @@ function readQuotaSettlementFromRequest( const writeback = writebackResult(identity, writebackRun, writebackEvent); const spend = spendResult(identity, spendRun, spendEvent); + const semanticReplanGuard = projectSemanticReplanGuard(receiptDetails); + const retiredNoSpend = writeback.failure === null && spendRun === null && spendEvent === null + && isCapabilityRetirementWriteback(writebackRun, identity, semanticReplanGuard.selected_capability_guard); // The exact Turn-bound blocked writeback is itself a durable no-spend // closeout. It cannot certify Todo completion or become delivery progress. // A spend already committed for this identity remains an ordinary spend @@ -1024,7 +1051,8 @@ function readQuotaSettlementFromRequest( ); const terminalCloseout = terminalResult(identity, completionEvent); const withWriteback = settlementBindReduce(identityResult, writeback); - const settled = blockedNoSpend ? withWriteback : settlementBindReduce(withWriteback, spend); + const noSpendKind = retiredNoSpend ? "capability_duty_retired_no_spend" : blockedNoSpend ? "typed_blocked_writeback_no_spend" : null; + const settled = noSpendKind ? withWriteback : settlementBindReduce(withWriteback, spend); const terminalSettlement = settlementBindReduce(settled, terminalCloseout); const monitorPoll = committedMonitorPollFromSnapshot(snapshot, identity); const nestedCausality = typeof receiptDetails.delivery_workspace_causality === "object" && @@ -1042,7 +1070,6 @@ function readQuotaSettlementFromRequest( const workspaceCausality: DeliveryWorkspaceCausality | null = normalizeDeliveryWorkspaceCausality(nestedCausality, identity.todo_id) ?? normalizeDeliveryWorkspaceCausality(flatCausality, identity.todo_id); - const semanticReplanGuard = projectSemanticReplanGuard(receiptDetails); const todoBoundReplan = identity.binding_kind === "todo" && semanticReplanGuard.scope === "turn_guard" && semanticReplanGuard.selected_obligation_id !== null; @@ -1056,7 +1083,7 @@ function readQuotaSettlementFromRequest( optionalString(supersedeEvent.todo_id) === identity.todo_id, durable_writeback_present: writeback.failure === null, quota_spend_present: spend.failure === null, - no_spend_closeout_present: blockedNoSpend, + no_spend_closeout_present: noSpendKind !== null, }); const recovery = request.refresh_retry === null ? null : refreshRecovery( @@ -1081,7 +1108,7 @@ function readQuotaSettlementFromRequest( terminal_closeout: bundle(terminalCloseout), terminal_settlement: bundle(terminalSettlement), progress: settlementProgress(identityResult, writeback, spend, writebackRun, spendRun, - receiptDetails.quota_spend_source ?? spendRun?.source, blockedNoSpend), + receiptDetails.quota_spend_source ?? spendRun?.source, noSpendKind), workspace_causality: workspaceCausality, semantic_replan_guard: semanticReplanGuard, writeback_run: writebackRun, diff --git a/loopx/control_plane/quota/should_run_packet.py b/loopx/control_plane/quota/should_run_packet.py index d7989d1559..217a1ce643 100644 --- a/loopx/control_plane/quota/should_run_packet.py +++ b/loopx/control_plane/quota/should_run_packet.py @@ -1435,9 +1435,8 @@ def _build_active_quota_payload( next_action_warning=route.next_action_warning, replan_obligation=prepared.replan_obligation, ) - bounded_research_frontier = _dict_field( - prepared.status_payload, "bounded_research_frontier" - ) + frontier = _dict_field(prepared.status_payload, "bounded_research_frontier") + bounded_research_frontier = _dict_field(frontier or {}, "public_projection") or frontier _attach_truthy_fields( payload, bounded_research_frontier=bounded_research_frontier, @@ -1561,7 +1560,8 @@ def _build_settled_quota_payload( missing_gates=prepared.item.get("missing_gates"), agent_todo_summary=compact_quota_todo_summary_for_payload(prepared.agent_todo_summary) if prepared.agent_todo_summary else None, user_todo_summary=compact_quota_todo_summary_for_payload(prepared.user_todo_summary) if prepared.user_todo_summary else None, - bounded_research_frontier=_dict_field(prepared.status_payload, "bounded_research_frontier"), + bounded_research_frontier=(_dict_field(_dict_field(prepared.status_payload, "bounded_research_frontier") or {}, "public_projection") + or _dict_field(prepared.status_payload, "bounded_research_frontier")), ) if prepared.agent_scoped_user_todo_override: payload[str(prepared.agent_scoped_user_todo_override["kind"])] = prepared.agent_scoped_user_todo_override diff --git a/loopx/control_plane/quota/slot_accounting.py b/loopx/control_plane/quota/slot_accounting.py index 387ed81304..1f5e1fc77e 100644 --- a/loopx/control_plane/quota/slot_accounting.py +++ b/loopx/control_plane/quota/slot_accounting.py @@ -182,13 +182,15 @@ def _resolve_preview_settlement( if ( isinstance(readback.progress, dict) and readback.progress.get("closeout_kind") - == "typed_blocked_writeback_no_spend" + in {"typed_blocked_writeback_no_spend", "capability_duty_retired_no_spend"} ): return { "identity": readback.identity.value, "result": readback.settlement, "delivery_run": readback.writeback_run, "reason": ( + "this exact capability duty was retired and must not consume a quota slot; reassess the current frontier in a new Turn" + if readback.progress.get("closeout_kind") == "capability_duty_retired_no_spend" else "this Turn already closed with an exact typed blocked writeback " "and must not consume a quota slot; retry the Todo only after " "its external blocker changes or a bounded backoff" diff --git a/loopx/control_plane/quota/unsettled_host_turn_recovery.ts b/loopx/control_plane/quota/unsettled_host_turn_recovery.ts index c892b37277..0601c52091 100644 --- a/loopx/control_plane/quota/unsettled_host_turn_recovery.ts +++ b/loopx/control_plane/quota/unsettled_host_turn_recovery.ts @@ -63,6 +63,7 @@ const MISSING_RECEIPT_NAMES = [WRITEBACK_RECEIPT, SPEND_RECEIPT] as const; export const ACCEPTED_CLOSEOUTS = [ "validated_writeback_and_quota_spend", "typed_blocked_writeback_no_spend", + "capability_duty_retired_no_spend", "exact_committed_quota_monitor_poll", "typed_external_wait_with_runnable_successor", "typed_blocker_or_lifecycle_transition", @@ -255,8 +256,8 @@ export async function preflightPriorHostTurnCloseout( newestSettledTurn ??= selected.prior_turn_instance_id; if (newestSettledTurn === selected.prior_turn_instance_id) { const progress = jsonObject(readback.progress); - if (progress?.closeout_kind === "typed_blocked_writeback_no_spend") { - newestAcceptedCloseout = "typed_blocked_writeback_no_spend"; + if (progress?.closeout_kind === "typed_blocked_writeback_no_spend" || progress?.closeout_kind === "capability_duty_retired_no_spend") { + newestAcceptedCloseout = progress.closeout_kind; } } continue; diff --git a/loopx/control_plane/status/agent_lane_projection.py b/loopx/control_plane/status/agent_lane_projection.py index 88cf37c9f7..58d7fb965c 100644 --- a/loopx/control_plane/status/agent_lane_projection.py +++ b/loopx/control_plane/status/agent_lane_projection.py @@ -30,6 +30,7 @@ "agent_reward_memory", "autonomous_replan_ack", "autonomous_replan_obligation", + "bounded_research_frontier", "completed_todo_archive_warning", "control_plane", "external_progress_review", diff --git a/loopx/control_plane/work_items/progress_observation.py b/loopx/control_plane/work_items/progress_observation.py index 618b2dd0e3..871ebcd7d5 100644 --- a/loopx/control_plane/work_items/progress_observation.py +++ b/loopx/control_plane/work_items/progress_observation.py @@ -645,6 +645,9 @@ def build_replan_action_packet( ProgressResultClass.NO_FOLLOWUP.value, ], } + if isinstance(obligation.get("capability_guard"), Mapping): + packet["capability_guard"] = dict(obligation["capability_guard"]) + packet["allowed_terminal"] = [] if isinstance(selected_gap, Mapping): packet["bounded_frontier"] = { key: selected_gap[key] @@ -654,6 +657,8 @@ def build_replan_action_packet( "experiment_node_ref", "input_node_refs", "required_outcome", + "input_observations", + "frontier_revision", ) if key in selected_gap } diff --git a/loopx/control_plane/work_items/replan_semantics.ts b/loopx/control_plane/work_items/replan_semantics.ts index 36371e6df3..7062eab8cc 100644 --- a/loopx/control_plane/work_items/replan_semantics.ts +++ b/loopx/control_plane/work_items/replan_semantics.ts @@ -11,8 +11,10 @@ const VISION_OUTCOMES = [ "fresh_vision_path_outcome", "new_runnable_successor", "new_concrete_blocker", "coverage_backed_exploration_exhausted", "coverage_backed_no_followup", ] as const; -type SemanticOutcome = typeof PROGRESS_OUTCOMES[number] | typeof VISION_OUTCOMES[number]; -const KNOWN_OUTCOMES: ReadonlySet = new Set([...PROGRESS_OUTCOMES, ...VISION_OUTCOMES]); +// Capability-owned evidence is never a default generic progress exit. +const CAPABILITY_OUTCOMES = ["capability_evidence_observed", "capability_duty_retired"] as const; +type SemanticOutcome = typeof PROGRESS_OUTCOMES[number] | typeof VISION_OUTCOMES[number] | typeof CAPABILITY_OUTCOMES[number]; +const KNOWN_OUTCOMES: ReadonlySet = new Set([...PROGRESS_OUTCOMES, ...VISION_OUTCOMES, ...CAPABILITY_OUTCOMES]); // Outcomes that a renamed identifier alone can produce. An external progress // review found the evaluated identifiers not serving the goal, so for that // source they discharge only behind evidence ids absent from the whole @@ -151,6 +153,10 @@ export function requiredSemanticOutcomes(obligation: JsonObject): SemanticOutcom if (declared.some(value => !KNOWN_OUTCOMES.has(value))) { throw new EffectRuntimeRequestError("satisfying_semantic_outcomes contains an unknown typed outcome"); } + if (declared.some(value => (CAPABILITY_OUTCOMES as readonly string[]).includes(value)) && + object(obligation.capability_guard).schema_version !== "semantic_replan_capability_guard_v0") { + throw new EffectRuntimeRequestError("capability evidence requires a bound owning capability"); + } if (acceptanceHold && declared.some(value => !["new_runnable_successor", "new_concrete_blocker"].includes(value))) { throw new EffectRuntimeRequestError("acceptance recovery cannot widen its typed outcomes"); } @@ -220,6 +226,9 @@ export function projectReplanSemantics(value: unknown): JsonObject { const patch = object(vision.vision_patch); const path = object(vision.path_delta); let outcomes = strings(observation.delta_kinds); + if (outcomes.some(value => (CAPABILITY_OUTCOMES as readonly string[]).includes(value))) { + throw new EffectRuntimeRequestError("capability evidence must be qualified by its owning capability, not generic progress"); + } if (outcomes.some(outcome => !KNOWN_OUTCOMES.has(outcome))) { throw new EffectRuntimeRequestError("observation_delta contains an unknown typed outcome"); } diff --git a/loopx/control_plane/work_items/semantic_replan_writeback.py b/loopx/control_plane/work_items/semantic_replan_writeback.py index 0f7047df9c..85051c8e14 100644 --- a/loopx/control_plane/work_items/semantic_replan_writeback.py +++ b/loopx/control_plane/work_items/semantic_replan_writeback.py @@ -3,7 +3,7 @@ from __future__ import annotations import shlex -from collections.abc import Mapping +from collections.abc import Callable, Mapping from dataclasses import dataclass from typing import Any @@ -42,6 +42,14 @@ REPLAN_WRITEBACK_REJECTION_SCHEMA_VERSION = "replan_writeback_rejection_v0" +@dataclass(frozen=True) +class CapabilityReplanEvidence: + """Composition-root supplied facts and their capability-owned qualifier.""" + frontier: dict[str, Any] + qualify: Callable[[Mapping[str, Any], str, dict[str, Any] | None, list[dict[str, Any]]], + tuple[dict[str, Any], dict[str, Any]]] + + @dataclass(frozen=True) class RefreshReplanQualification: repair_delta_contract: dict[str, Any] | None @@ -119,19 +127,26 @@ def project_replan_writeback_rejection( runtime_root=runtime_root, ) ) + capability_guard = obligation.get("capability_guard") + if isinstance(capability_guard, Mapping) and rejection.semantic_delta.get("readback_actions"): + next_cli_actions = rejection.semantic_delta["readback_actions"] return { "schema_version": REPLAN_WRITEBACK_REJECTION_SCHEMA_VERSION, "required": True, "host_action": ( "settle_todo_lifecycle" if lifecycle_reentry is not None + else "read_current_capability_evidence" if capability_guard else "write_typed_semantic_delta" ), "obligation_id": obligation.get("obligation_id"), "resolution_mode": obligation.get("resolution_mode"), + **({"retirement_contract": rejection.semantic_delta["retirement_contract"]} + if rejection.semantic_delta.get("retirement_contract") else {}), "reason_code": rejection.semantic_delta.get("reason_code"), "triggers": triggers, "next_cli_actions": next_cli_actions, + **({"capability_guard": dict(capability_guard)} if isinstance(capability_guard, Mapping) else {}), } @@ -200,6 +215,8 @@ def qualify_replan_writeback( external_progress_review: Mapping[str, Any] | None = None, guard_scoped: bool = False, guard_semantic_replan_obligation_id: str | None = None, + capability_evidence: CapabilityReplanEvidence | None = None, + guard_capability: Mapping[str, Any] | None = None, ) -> tuple[dict[str, Any] | None, dict[str, Any] | None]: """Return the shared open obligation and the writeback's typed delta. @@ -261,12 +278,15 @@ def qualify_replan_writeback( "run_history": { "goals": [ { + **(registry_goal or {}), "id": goal_id, "latest_runs": list(newest_first_runs or []), } ] } } + if capability_evidence is not None: + status_payload["bounded_research_frontier"] = capability_evidence.frontier context = build_goal_frontier_projection_context_from_status( goal_id=goal_id, agent_id=safe_agent_id, @@ -297,6 +317,14 @@ def qualify_replan_writeback( else None ), ) + obligation = context.get("replan_obligation") + selected_capability = guard_capability if guard_scoped else (obligation or {}).get("capability_guard") + if selected_capability is not None: + if capability_evidence is None: + raise ValueError("selected capability writeback owner is not available") + selected_id = guard_semantic_replan_obligation_id if guard_scoped else (obligation or {}).get("obligation_id") + return capability_evidence.qualify(selected_capability, str(selected_id), progress_observation, + [run for run in newest_first_runs or [] if run.get("agent_id") == safe_agent_id]) if guard_scoped and guard_semantic_replan_obligation_id: transition_delta = guarded_replan_transition_delta( guard_scoped=guard_scoped, @@ -364,6 +392,8 @@ def enforce_open_replan_writeback( guard_semantic_replan_obligation_id: str | None = None, todo_fields: dict[str, Any] | None = None, external_progress_review: Mapping[str, Any] | None = None, + capability_evidence: CapabilityReplanEvidence | None = None, + guard_capability: Mapping[str, Any] | None = None, ) -> dict[str, Any] | None: """Fail closed unless concrete typed evidence satisfies the selected replan. @@ -388,6 +418,7 @@ def enforce_open_replan_writeback( todo_fields=todo_fields, guard_scoped=guard_scoped, guard_semantic_replan_obligation_id=guard_semantic_replan_obligation_id, + capability_evidence=capability_evidence, guard_capability=guard_capability, ) if not obligation: if isinstance(semantic_delta, dict) and semantic_delta.get("accepted") is True: @@ -444,6 +475,8 @@ def qualify_refresh_replan_writeback( classification: str, delivery_outcome: str | None, todo_fields: dict[str, Any] | None = None, + capability_evidence: CapabilityReplanEvidence | None = None, + guard_capability: Mapping[str, Any] | None = None, ) -> RefreshReplanQualification: """Qualify one refresh's replan delta and accountable settlement outcome.""" @@ -495,6 +528,7 @@ def qualify_refresh_replan_writeback( ) semantic_delta = enforce_open_replan_writeback( + capability_evidence=capability_evidence, guard_capability=guard_capability, newest_first_runs=newest_first_runs, state_text=state_text, agent_id=agent_id, @@ -512,6 +546,8 @@ def qualify_refresh_replan_writeback( ), ) if semantic_delta: + if semantic_delta.get("retirement") and delivery_outcome != "outcome_gap": + raise ValueError("capability duty retirement requires outcome_gap; it is not delivery progress") effective_recorded = True classification = requested_classification delivery_outcome = requested_delivery_outcome diff --git a/loopx/control_plane/work_items/task_lease_workspace.ts b/loopx/control_plane/work_items/task_lease_workspace.ts index 3c6778dafe..6f0b75b5d2 100644 --- a/loopx/control_plane/work_items/task_lease_workspace.ts +++ b/loopx/control_plane/work_items/task_lease_workspace.ts @@ -5,11 +5,11 @@ import {realpath, stat, readFile, lstat, opendir} from "node:fs/promises"; import {platform} from "node:os"; import {createHash} from "node:crypto"; import {isAbsolute, resolve} from "node:path"; -import {BARE_SHA256_PATTERN} from "../content_digest.ts"; import type {JsonObject} from "../effect_program.ts"; import {EffectRuntimeRequestError} from "../effect_runtime_errors.ts"; import {requireJsonObject} from "../runtime_decode.ts"; import {leaseWriteRepository} from "./task_lease_repository.ts"; +import {BARE_SHA256_PATTERN} from "../content_digest.ts"; export interface LeaseWorkspace extends JsonObject { host: string; diff --git a/loopx/orchestration.py b/loopx/orchestration.py index 76c060d5ba..7bdaad2e31 100644 --- a/loopx/orchestration.py +++ b/loopx/orchestration.py @@ -128,6 +128,12 @@ def compact_explore_harness_policy(policy: Any) -> dict[str, Any]: profile = str(harness.get("profile") or "").strip() if profile: compact["profile"] = profile + if "composition_mode" in harness or "composition_scope_id" in harness: + from .capabilities.explore.research_evidence import _research_result + policy = _research_result("explore.research.composition_policy", {"harness": harness}) + compact["composition_mode"] = policy["mode"] + if policy["coverage_scope_id"] is not None: + compact["composition_scope_id"] = policy["coverage_scope_id"] return compact diff --git a/loopx/presentation/renderers/status_markdown.py b/loopx/presentation/renderers/status_markdown.py index 7ab021b9a5..5832d4f226 100644 --- a/loopx/presentation/renderers/status_markdown.py +++ b/loopx/presentation/renderers/status_markdown.py @@ -629,6 +629,15 @@ def append_project_asset_warning_markdown( f"trigger_count={replan_obligation.get('trigger_count')} " f"triggers={markdown_scalar(','.join(trigger_kinds))}" ) + research = as_dict(project_asset.get("bounded_research_frontier")) + if research: + lines.append(" - research_execution: " + f"state={markdown_scalar(research.get('state'))} " + f"pending={research.get('pending_count')} scheduled={research.get('scheduled_count')} " + f"observed={research.get('observed_count')} ineligible={research.get('ineligible_count')} " + f"dismissed={research.get('dismissed_count')} deferred={research.get('deferred_count')}") + for gap in research.get("gaps") or []: + lines.append(f" - {markdown_scalar(gap.get('gap_id'))}: {markdown_scalar(gap.get('status'))}") interface_budget_cadence = ( project_asset.get("interface_budget_cadence") if isinstance(project_asset.get("interface_budget_cadence"), dict) diff --git a/loopx/quota.py b/loopx/quota.py index eb90263190..ced4333a46 100644 --- a/loopx/quota.py +++ b/loopx/quota.py @@ -1533,7 +1533,13 @@ def spend_quota_slot( "turn_instance_id": identity.turn_instance_id, "settlement_identity": identity.as_dict(), "settlement_result": settlement_result_payload(spent_result), - "reason": "quota spend receipt replayed for the same settlement identity", + **({"settlement_progress": settlement_readback.progress} + if settlement_readback.progress.get("closeout_kind") == "capability_duty_retired_no_spend" else {}), + "reason": ( + "exact Turn closeout replayed without a quota debit" + if settlement_readback.progress.get("closeout_kind") == "capability_duty_retired_no_spend" else + "quota spend receipt replayed for the same settlement identity" + ), } prior_spend_run = settlement_readback.spend_run if prior_spend_run is not None: diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index 767ddb1777..4dcca23032 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -165,6 +165,22 @@ "api": "load_registry", "classification": "codec_api" }, + { + "site": "loopx/capabilities/explore/research_evidence.py::.append_research_observation::codec_read:load_registry#1", + "line": 113, + "column": 39, + "kind": "codec_read", + "api": "load_registry", + "classification": "codec_api" + }, + { + "site": "loopx/capabilities/explore/research_frontier.py::.hold_research_completion_evidence::codec_read:load_registry#1", + "line": 199, + "column": 31, + "kind": "codec_read", + "api": "load_registry", + "classification": "codec_api" + }, { "site": "loopx/capabilities/issue_fix/explore_projection.py::.project_issue_fix_explore_graph::codec_read:load_registry#1", "line": 643, @@ -637,9 +653,17 @@ "api": "load_registry", "classification": "codec_api" }, + { + "site": "loopx/cli_commands/explore.py::._projection_for::codec_read:load_registry#1", + "line": 271, + "column": 20, + "kind": "codec_read", + "api": "load_registry", + "classification": "codec_api" + }, { "site": "loopx/cli_commands/explore.py::.handle_explore_command::codec_read:load_registry#1", - "line": 466, + "line": 486, "column": 20, "kind": "codec_read", "api": "load_registry", @@ -743,7 +767,7 @@ }, { "site": "loopx/cli_commands/project_lifecycle_refresh_state.py::.handle_refresh_state_command::codec_read:load_registry#1", - "line": 560, + "line": 561, "column": 17, "kind": "codec_read", "api": "load_registry", @@ -751,7 +775,7 @@ }, { "site": "loopx/cli_commands/project_lifecycle_refresh_state.py::.handle_refresh_state_command::codec_read:load_registry#2", - "line": 639, + "line": 640, "column": 17, "kind": "codec_read", "api": "load_registry", @@ -855,7 +879,7 @@ }, { "site": "loopx/cli_commands/todo.py::._validated_replan_successor_obligation::codec_read:load_registry#1", - "line": 178, + "line": 184, "column": 16, "kind": "codec_read", "api": "load_registry", @@ -863,7 +887,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#1", - "line": 309, + "line": 330, "column": 49, "kind": "codec_read", "api": "load_registry", @@ -871,7 +895,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#2", - "line": 325, + "line": 346, "column": 24, "kind": "codec_read", "api": "load_registry", @@ -879,7 +903,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#3", - "line": 333, + "line": 354, "column": 24, "kind": "codec_read", "api": "load_registry", @@ -887,7 +911,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#4", - "line": 513, + "line": 534, "column": 53, "kind": "codec_read", "api": "load_registry", @@ -895,7 +919,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#5", - "line": 634, + "line": 655, "column": 61, "kind": "codec_read", "api": "load_registry", @@ -903,7 +927,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#6", - "line": 698, + "line": 719, "column": 13, "kind": "codec_read", "api": "load_registry", @@ -911,7 +935,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#7", - "line": 740, + "line": 761, "column": 38, "kind": "codec_read", "api": "load_registry", @@ -935,7 +959,7 @@ }, { "site": "loopx/configure_goal.py::.configure_goal::codec_transaction:project_registry_transaction#1", - "line": 515, + "line": 517, "column": 14, "kind": "codec_transaction", "api": "project_registry_transaction", @@ -943,7 +967,7 @@ }, { "site": "loopx/configure_goal.py::.configure_goal::codec_read:load_project_registry#1", - "line": 725, + "line": 727, "column": 14, "kind": "codec_read", "api": "load_project_registry", diff --git a/loopx/state_refresh.py b/loopx/state_refresh.py index 02d7f10deb..cecd7881d1 100644 --- a/loopx/state_refresh.py +++ b/loopx/state_refresh.py @@ -903,7 +903,7 @@ def refresh_state_run( # Only pure input validation runs before this transitional persistence lock. with (nullcontext() if dry_run else exclusive_run_index_lock( runtime_root / "goals" / safe_goal_id / "runs" / "index.jsonl", operation="refresh-state" - )): + )), ExitStack() as research_write_guard: settlement_identity = None settlement_result = None delivery_workspace_causality = None @@ -1004,6 +1004,14 @@ def refresh_state_run( project_override=project, state_file_override=state_file, ) + harness = ((registry_goal or {}).get("spawn_policy") or {}).get("explore_harness") or {} + capability_guard = (getattr(settlement_readback, "semantic_replan_guard", None) or {}).get("selected_capability_guard") + if (harness.get("enabled") is True and harness.get("composition_mode") == "explicit_only") or capability_guard is not None: + from .capabilities.explore.result_log import explore_result_log_path + research_write_guard.enter_context(exclusive_file_lock( + explore_result_log_path(runtime_root, safe_goal_id), + agent_id=normalized_agent_id or None, operation="research-writeback", + )) planning_source = load_refresh_planning_source( runtime_root, safe_goal_id, resolved_state_file, require_display=bool(next_action) ) @@ -1132,7 +1140,13 @@ def refresh_state_run( and settlement_readback.semantic_replan_guard is not None else {} ) + from .capabilities.explore.research_frontier import prepare_research_replan_evidence + capability_evidence = prepare_research_replan_evidence(runtime_root=runtime_root, goal_id=safe_goal_id, + agent_id=normalized_agent_id, registry_goal=registry_goal, state_text=state_text, + capability_guard=settlement_replan_guard.get("selected_capability_guard")) replan_qualification = qualify_refresh_replan_writeback( + capability_evidence=capability_evidence, + guard_capability=settlement_replan_guard.get("selected_capability_guard"), todo_fields=todo_fields, autonomous_replan_recorded=autonomous_replan_recorded, requested_delta_kinds=normalized_repair_delta_kinds, diff --git a/loopx/todos.py b/loopx/todos.py index b015a2015e..8cc030eb87 100644 --- a/loopx/todos.py +++ b/loopx/todos.py @@ -1714,6 +1714,12 @@ def complete_goal_todo( ) ) completion_state = completion_transaction.get("completion_state") + from .capabilities.explore.research_frontier import hold_research_completion_evidence + research_evidence = lease_fence_stack.enter_context(hold_research_completion_evidence( + registry_path=registry_path, runtime_root=shadow_runtime_root, goal_id=goal_id, + todo=completion_todo, state_text=original, + actor_agent_id=mutation_authority.get("actor_agent_id"), + )) completion_policy = completion_policy_from_transaction(completion_transaction) effective_claimed_by = completion_policy.effective_claimed_by registered_agents = completion_policy.registered_agents @@ -1830,6 +1836,7 @@ def complete_goal_todo( ) result = { "ok": True, + **({"capability_completion_evidence": research_evidence} if research_evidence is not None else {}), "dry_run": dry_run, "completed": True, "goal_id": goal_id, @@ -1930,6 +1937,11 @@ def supersede_goal_todo( idempotency_key=task_lease_idempotency_key, expected_version=task_lease_expected_version, runtime_root=shadow_runtime_root, ) + from .capabilities.explore.research_frontier import hold_research_completion_evidence + research_completion_evidence = lease_fence_stack.enter_context(hold_research_completion_evidence( + registry_path=registry_path, runtime_root=shadow_runtime_root, goal_id=goal_id, + todo=authority_todo, state_text=original, actor_agent_id=mutation_authority.get("actor_agent_id"), + )) update_result = apply_todo_update_to_lines( lines, todo_id=todo_id, @@ -2001,6 +2013,7 @@ def supersede_goal_todo( "ok": True, "dry_run": dry_run, "superseded": True, + **({"capability_completion_evidence": research_completion_evidence} if research_completion_evidence else {}), "goal_id": goal_id, **update_result, "changed": changed, diff --git a/tests/architecture/test_semantic_development_probe.py b/tests/architecture/test_semantic_development_probe.py index 94987444c4..ec49dc2ef1 100644 --- a/tests/architecture/test_semantic_development_probe.py +++ b/tests/architecture/test_semantic_development_probe.py @@ -2,6 +2,7 @@ from __future__ import annotations +import os from pathlib import Path import subprocess import sys @@ -344,6 +345,13 @@ def probe_cli(repository: Path) -> Path: def _run_probe_cli(repository: Path) -> subprocess.CompletedProcess[str]: + # The child executes copied sources under a disposable loopx/ package. Do + # not attribute those temporary files to the checkout's coverage artifact: + # shard artifacts are combined after pytest has removed the temp checkout. + env = { + key: value for key, value in os.environ.items() + if not key.startswith("COV_CORE_") and key != "COVERAGE_PROCESS_START" + } return subprocess.run( [ sys.executable, @@ -352,6 +360,7 @@ def _run_probe_cli(repository: Path) -> subprocess.CompletedProcess[str]: "HEAD", ], cwd=repository, + env=env, capture_output=True, text=True, check=False, diff --git a/tests/architecture/test_turn_contract_generation.py b/tests/architecture/test_turn_contract_generation.py index 9bd8cf1e28..0fcc1c023b 100644 --- a/tests/architecture/test_turn_contract_generation.py +++ b/tests/architecture/test_turn_contract_generation.py @@ -260,9 +260,12 @@ def test_new_independent_twin_cannot_hide_behind_generated_pair(monkeypatch): ) assert counts is not None raw, generated, maintained, budget = map(int, counts.groups()) - # Source-verified bindings may grow as decision owners converge. The - # independently maintained twin budget remains the frozen limit below. - assert generated >= 1 + # Require both reviewed generators while allowing later verified pairs. + from scripts.generate_semantic_bindings import verified_generated_paths as semantic_generated_paths + + assert "loopx/control_plane/turn_driver/turn_contract_generated.py" in generator.verified_generated_paths() + assert "loopx/control_plane/content_digest.py" in semantic_generated_paths() + assert generated >= 2 assert raw == maintained + generated from loopx.semantics.inventory import SourceFile diff --git a/tests/capabilities/test_explore_research_evidence.py b/tests/capabilities/test_explore_research_evidence.py index e1d323b359..60373e2c99 100644 --- a/tests/capabilities/test_explore_research_evidence.py +++ b/tests/capabilities/test_explore_research_evidence.py @@ -13,6 +13,7 @@ build_explore_result_projection, explore_result_log_path, load_explore_result_events_strict, ) from loopx.extensions.lark.presentation.explore_results import _node_record_values +from loopx.todos import add_goal_todo, update_goal_todo GOAL = "research-fixture" @@ -176,3 +177,109 @@ def command(*args: str) -> dict: view = command("summary", "--goal-id", GOAL) assert view["nodes"][0]["research_observation"] == result["observation"] assert view["research_frontier"]["mode"] == "read_only_shadow" + + +def execution_fixture(tmp_path: Path) -> tuple[Path, Path, Path, dict]: + project = tmp_path / "project" + project.mkdir() + state = project / "ACTIVE_GOAL_STATE.md" + state.write_text(f"---\ngoal_id: {GOAL}\n---\n\n## Agent Todo\n\n") + runtime = tmp_path / "runtime" + registry = tmp_path / "registry.json" + registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{ + "id": GOAL, "repo": str(project), "state_file": state.name, "status": "active", + "coordination": {"agent_model": "peer_v1", "registered_agents": ["fixture-agent"]}, + }]})) + path = explore_result_log_path(runtime, GOAL) + for name in ["a", "b"]: + node(path, name) + append_research_observation(path, goal_id=GOAL, observation=observation("b")) + append_research_observation(path, goal_id=GOAL, observation=observation("a", target="b")) + node(path, "joint", kind="experiment") + for name in ["a", "b"]: + append_explore_result_event(path, build_explore_edge_event( + goal_id=GOAL, from_node="joint", to_node=name, edge_type="depends_on")) + todo = add_goal_todo( + registry_path=registry, goal_id=GOAL, role="agent", text="Run the bounded joint experiment.", + task_class="advancement_task", action_kind="joint_probe", claimed_by="fixture-agent", + explore_result_node_refs=["joint"], monitor_metadata={"target_key": "joint"}, + replan_obligation_id="replan-0123456789abcdef", + ) + gap = projection(path)["research_frontier"]["gaps"][0] + raw = observation("joint") + raw["progress"]["work_item_id"] = todo["todo_id"] + raw["input_observations"] = gap["input_observations"] + raw["execution_lineage"] = { + "schema_version": "research_execution_lineage_v0", "goal_id": GOAL, + "gap_id": gap["gap_id"], "replan_obligation_id": "replan-0123456789abcdef", + "successor_todo_id": todo["todo_id"], "agent_id": "fixture-agent", + } + return registry, runtime, path, raw + + +def test_real_execution_writer_reads_todo_and_replay_does_not_rebind(tmp_path: Path) -> None: + registry, runtime, path, raw = execution_fixture(tmp_path) + update_goal_todo( + registry_path=registry, goal_id=GOAL, todo_id=raw["execution_lineage"]["successor_todo_id"], + agent_id="fixture-agent", status="deferred", reason="await-synthetic-capacity", + resume_when="capacity_available:fixture", + ) + before = path.read_bytes() + with pytest.raises(ValueError, match="runnable joint-probe"): + append_research_observation(path, goal_id=GOAL, observation=raw, agent_id="fixture-agent", + registry_path=registry, runtime_root=runtime) + assert path.read_bytes() == before + update_goal_todo( + registry_path=registry, goal_id=GOAL, todo_id=raw["execution_lineage"]["successor_todo_id"], + agent_id="fixture-agent", status="open", clear_resume_when=True, + ) + result = append_research_observation(path, goal_id=GOAL, observation=raw, agent_id="fixture-agent", + registry_path=registry, runtime_root=runtime) + assert result["written"] + assert result["observation"]["execution_lineage"] == raw["execution_lineage"] + # A historical receipt remains read-only if current task or input authority + # changes. It cannot be refreshed into a new execution claim. + update_goal_todo( + registry_path=registry, goal_id=GOAL, todo_id=raw["execution_lineage"]["successor_todo_id"], + agent_id="fixture-agent", action_kind="inspect", + ) + node(path, "a", status="open") + before = path.read_bytes() + assert append_research_observation(path, goal_id=GOAL, observation=raw)["replayed"] + assert path.read_bytes() == before + assert projection(path)["research_frontier"]["observed_count"] == 0 + + +@pytest.mark.parametrize("batch", [False, True]) +def test_generic_writers_cannot_forge_new_execution_lineage(tmp_path: Path, batch: bool) -> None: + _registry, _runtime, path, raw = execution_fixture(tmp_path) + event = build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="resolved", research_observation=raw) + before = path.read_bytes() + with pytest.raises(ValueError, match="requires explore observe"): + if batch: + append_explore_result_events(path, [event], expected_goal_id=GOAL) + else: + append_explore_result_event(path, event) + assert path.read_bytes() == before + + +def test_cli_execution_requires_actor_and_preserves_lineage_in_readback(tmp_path: Path) -> None: + registry, runtime, path, raw = execution_fixture(tmp_path) + packet = tmp_path / "execution.json" + packet.write_text(json.dumps(raw)) + command = [sys.executable, "-m", "loopx.cli", "--registry", str(registry), "--format", "json", + "explore", "observe", "--goal-id", GOAL, "--observation-json", str(packet)] + before = path.read_bytes() + missing = subprocess.run(command, capture_output=True, text=True, check=False) + assert missing.returncode != 0 + assert "actor differs" in missing.stdout + assert path.read_bytes() == before + accepted = subprocess.run([*command, "--agent-id", "fixture-agent"], capture_output=True, text=True, check=False) + assert accepted.returncode == 0, accepted.stdout + accepted.stderr + result = json.loads(accepted.stdout) + assert result["observation"]["execution_lineage"] == raw["execution_lineage"] + readback = projection(path) + joint = next(row for row in readback["nodes"] if row["node_id"] == "joint") + assert joint["research_observation"] == result["observation"] + assert readback["research_frontier"]["observed_count"] == 1 diff --git a/tests/capabilities/test_research_composition_gate.py b/tests/capabilities/test_research_composition_gate.py new file mode 100644 index 0000000000..bbfa7a4379 --- /dev/null +++ b/tests/capabilities/test_research_composition_gate.py @@ -0,0 +1,404 @@ +"""Real CLI admission, successor and writeback paths with synthetic evidence.""" +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +from loopx.capabilities.explore.research_evidence import append_research_observation +from loopx.capabilities.explore.result_log import ( + append_explore_result_event, build_explore_edge_event, build_explore_node_event, explore_result_log_path, +) +from loopx.control_plane.work_items.semantic_replan_writeback import qualify_replan_writeback +from loopx.capabilities.explore.research_frontier import prepare_research_replan_evidence + +GOAL = "research-gate-fixture" +AGENT = "fixture-agent" + + +def observation(node: str, target: str | None = None) -> dict: + return {"schema_version": "typed_research_observation_v0", "explore_node_id": node, + "progress": {"schema_version": "typed_progress_observation_v0", "work_item_id": f"todo_{node}", + "result_class": "exploration_exhausted", "coverage_scope_id": f"scope-{node}", + "coverage_complete": True, "evidence_ids": [f"ev-{node}"]}, + "closure_basis": {"schema_version": "research_closure_basis_v0", "disposition": "bounded", + "constraints": [{"kind": "invariant", "id": "boundary", "role": "decisive"}], + "evidence_ids": [f"ev-{node}"]}, + "composition_candidates": [{"basis": "explicit", "target_node_id": target, + "interaction_kind": "state_interference", "evidence_ids": [f"ev-{node}", f"ev-{target}"]}] if target else []} + + +@pytest.fixture +def fixture(tmp_path: Path): + state = tmp_path / "ACTIVE_GOAL_STATE.md" + state.write_text('---\nstatus: active\nowner_mode: goal\nobjective: "Test the explicit research boundary."\n' + 'updated_at: 2026-09-28T00:00:00Z\n---\n\n# Research fixture\n\n' + '## Objective\n\nTest the explicit research boundary.\n\n## Next Action\n\nInspect the current evidence.\n\n' + '## Agent Todo\n\n- [ ] Observe the synthetic fixture.\n' + ' \n') + runtime, registry = tmp_path / "runtime", tmp_path / "registry.json" + registry.write_text(json.dumps({"schema_version": "0.1", "common_runtime_root": str(runtime), "goals": [{ + "id": GOAL, "repo": str(tmp_path), "state_file": state.name, "status": "active", "domain": "research-fixture", + "adapter": {"kind": "fixture_connected_delivery_v0", "status": "connected-delivery"}, "authority_sources": [], + "quota": {"compute": 1.0, "window_hours": 24, "allowed_slots": 20}, + "coordination": {"agent_model": "peer_v1", "registered_agents": [AGENT]}, + "spawn_policy": {"explore_harness": {"enabled": True}}, + }]})) + log = explore_result_log_path(runtime, GOAL) + for name in ["a", "b", "joint"]: + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id=name, title=f"Research {name}", + node_kind="experiment" if name == "joint" else "hypothesis", status="open" if name == "joint" else "resolved")) + append_research_observation(log, goal_id=GOAL, observation=observation("b")) + append_research_observation(log, goal_id=GOAL, observation=observation("a", "b")) + for name in ["a", "b"]: + append_explore_result_event(log, build_explore_edge_event(goal_id=GOAL, from_node="joint", to_node=name, edge_type="depends_on")) + + def cli(*args: str, success: bool = True) -> dict: + result = subprocess.run([sys.executable, "-m", "loopx.cli", "--format", "json", "--registry", str(registry), + "--runtime-root", str(runtime), *args], capture_output=True, text=True, check=False, + cwd=Path(__file__).resolve().parents[2]) + assert (result.returncode == 0) is success, result.stdout + result.stderr + return json.loads(result.stdout) + + return cli, log, runtime, registry, state + + +def activate(cli) -> None: + assert cli("configure-goal", "--goal-id", GOAL, "--explore-composition-mode", "explicit_only", + "--explore-composition-scope-id", "scope-joint", "--execute")["ok"] + + +def test_inactive_policy_never_requires_a_research_runtime(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr("loopx.capabilities.explore.composition_frontier.project_live_explore_composition_frontier", + lambda **_kwargs: pytest.fail("disabled policy must not read research state")) + for harness in [{"enabled": True}, {"enabled": True, "composition_mode": "disabled"}, + {"enabled": False, "composition_mode": "explicit_only", "composition_scope_id": "scope"}]: + obligation, _ = qualify_replan_writeback(newest_first_runs=[], state_text="## Agent Todo\n\n", agent_id=AGENT, + goal_id=GOAL, registry_goal={"id": GOAL, "spawn_policy": {"explore_harness": harness}}) + assert obligation is None + + +def test_real_guard_and_successor_share_the_current_gap(fixture) -> None: + cli, log, runtime, registry, state = fixture + old = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-off") + assert old.get("effective_action") != "autonomous_replan_required" + assert "research_execution_frontier" not in cli("explore", "summary", "--goal-id", GOAL) + off_status = cli("status", "--goal-id", GOAL, "--agent-id", AGENT) + assert all("bounded_research_frontier" not in item.get("project_asset", {}) + for item in off_status["attention_queue"]["items"]) + activate(cli) + guard = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-on") + assert guard["effective_action"] == "autonomous_replan_required" + packet = guard["replan_action_packet"] + obligation = packet["obligation_id"] + assert packet["capability_guard"]["gap_id"] == guard["bounded_research_frontier"]["selected_gap"]["gap_id"] + status = cli("status", "--goal-id", GOAL, "--agent-id", AGENT) + item = next(item for item in status["attention_queue"]["items"] if item["goal_id"] == GOAL) + assert item["project_asset"]["bounded_research_frontier"] == guard["bounded_research_frontier"] + assert item["project_asset"]["autonomous_replan_obligation"]["obligation_id"] == obligation + summary = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT) + assert summary["research_execution_frontier"] == guard["bounded_research_frontier"] + from loopx.cli_commands.explore import render_explore_markdown + from loopx.presentation.renderers.status_markdown import render_status_markdown + assert packet["capability_guard"]["gap_id"] in render_explore_markdown(summary) + assert packet["capability_guard"]["gap_id"] in render_status_markdown(status) + assert "lineage_gaps" not in guard["bounded_research_frontier"] + assert "settlement_transitions" not in guard["bounded_research_frontier"] + rejected = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", obligation, "--turn-instance-id", "fixture-on", + "--classification", "bounded_fixture_probe", "--delivery-batch-scale", "implementation", + "--delivery-outcome", "outcome_progress", "--progress-result-class", "advanced", + "--progress-surface-id", "unrelated", "--progress-evidence-id", "ev-unrelated", success=False) + assert "evidence duty" in json.dumps(rejected) + base = ["todo", "add", "--goal-id", GOAL, "--role", "agent", "--task-class", "advancement_task", + "--action-kind", "joint_probe", "--claimed-by", AGENT, "--replan-obligation-id", obligation] + bad = cli(*base, "--text", "Unrelated work.", "--explore-result-node-ref", "a", success=False) + assert "bind one current binary experiment" in json.dumps(bad) + deferred = cli(*base, "--text", "Deferred joint experiment.", "--explore-result-node-ref", "joint", + "--status", "deferred", "--resume-when", "capacity_available:fixture", success=False) + assert "no deferral" in json.dumps(deferred) + created = cli(*base, "--text", "Run the bounded joint experiment.", "--explore-result-node-ref", "joint", "--target-key", "joint") + assert created["replan_transition"]["recorded"] + before_closeout = state.read_bytes() + missing = cli("todo", "complete", "--goal-id", GOAL, "--todo-id", created["todo_id"], + "--agent-id", AGENT, "--claimed-by", AGENT, "--no-follow-up", + "--evidence", "No experiment result recorded.", success=False) + assert "current typed experiment observation" in json.dumps(missing) + assert state.read_bytes() == before_closeout + retired = cli("todo", "supersede", "--goal-id", GOAL, "--todo-id", created["todo_id"], + "--agent-id", AGENT, "--reason", "Attempt to close through another verb.", success=False) + assert "current typed experiment observation" in json.dumps(retired) + assert state.read_bytes() == before_closeout + replay = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-on") + assert replay["heartbeat_receipt"]["semantic_replan_capability_guard"] == packet["capability_guard"] + assert replay["replan_action_packet"]["obligation_id"] == obligation + refresh = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", obligation, "--turn-instance-id", "fixture-on", + "--classification", "bounded_fixture_probe", "--delivery-batch-scale", "implementation", + "--delivery-outcome", "outcome_progress", "--vision-summary", "Test this explicit research question.", + "--vision-acceptance", "The typed joint experiment addresses the question.", + "--vision-replan-trigger", "Research result remains open.") + assert refresh["ok"] + spent = cli("quota", "spend-slot", "--goal-id", GOAL, "--agent-id", AGENT, "--slots", "1", "--source", "heartbeat", + "--execute", "--replan-obligation-id", obligation, "--turn-instance-id", "fixture-on") + assert spent["settlement_progress"]["state"] == "settled" + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="resolved")) + result = observation("joint") + result["progress"]["work_item_id"] = created["todo_id"] + result["input_observations"] = guard["bounded_research_frontier"]["selected_gap"]["input_observations"] + result["execution_lineage"] = {"schema_version": "research_execution_lineage_v0", "goal_id": GOAL, + "gap_id": packet["capability_guard"]["gap_id"], "replan_obligation_id": obligation, + "successor_todo_id": created["todo_id"], "agent_id": AGENT} + append_research_observation(log, goal_id=GOAL, observation=result, agent_id=AGENT, + registry_path=registry, runtime_root=runtime) + complete = cli("todo", "complete", "--goal-id", GOAL, "--todo-id", created["todo_id"], + "--agent-id", AGENT, "--claimed-by", AGENT, "--no-follow-up", "--evidence", "Typed result recorded.") + assert complete["completed"] + assert complete["capability_completion_evidence"]["experiment_node_id"] == "joint" + archived = cli("todo", "archive-completed", "--goal-id", GOAL, "--role", "agent", + "--max-active-done", "0", "--execute") + assert archived["moved_count"] == 1 + assert cli("todo", "list", "--goal-id", GOAL, "--todo-id", created["todo_id"])["todos"][0]["archive_state"] == "archive" + from loopx.capabilities.explore.composition_frontier import project_live_explore_composition_frontier + frontier = project_live_explore_composition_frontier(runtime_root=runtime, goal_id=GOAL, agent_id=AGENT, + status_payload={"run_history": {"goals": json.loads(registry.read_text())["goals"]}}) + assert frontier["observed_count"] == 1 + assert frontier["scheduled_count"] == 0 + summary = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT) + assert summary["research_execution_frontier"]["observed_count"] == 1 + assert "lineage_gaps" not in summary["research_execution_frontier"] + from loopx.extensions.lark.presentation.explore_results import _node_record_values + joint = next(node for node in summary["nodes"] if node["node_id"] == "joint") + lark = _node_record_values(joint, goal_id=GOAL, source_id="synthetic-source") + assert "observed" in lark["Summary"] and packet["capability_guard"]["gap_id"] in lark["Summary"] + + +def test_selected_guard_cannot_be_erased_by_input_invalidation(fixture) -> None: + cli, log, _runtime, _registry, _state = fixture + activate(cli) + guard = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-stale") + obligation = guard["replan_action_packet"]["obligation_id"] + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="a", title="Research a", status="open")) + rejected = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", obligation, "--turn-instance-id", "fixture-stale", + "--classification", "bounded_fixture_probe", "--delivery-batch-scale", "implementation", + "--delivery-outcome", "outcome_progress", "--progress-result-class", "advanced", + "--progress-evidence-id", "ev-unrelated", success=False) + assert "evidence duty" in json.dumps(rejected) + + +def test_live_presentation_requires_explicit_registered_actor_for_multi_agent_goal(fixture) -> None: + cli, log, _runtime, registry, state = fixture + activate(cli) + source = json.loads(registry.read_text()) + source["goals"][0]["coordination"]["registered_agents"].append("other-agent") + registry.write_text(json.dumps(source)) + before = log.read_bytes(), state.read_bytes(), registry.read_bytes() + for args in [[], ["--agent-id", "unregistered"]]: + rejected = cli("explore", "summary", "--goal-id", GOAL, *args, success=False) + assert "registered Goal agent" in json.dumps(rejected) + summary = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT) + assert summary["research_execution_frontier"]["agent_id"] == AGENT + assert summary["research_execution_frontier"]["grants_execution_authority"] is False + assert (log.read_bytes(), state.read_bytes(), registry.read_bytes()) == before + + +def test_real_replan_guard_accepts_exact_blocker_wait_and_resumes_without_closure(fixture) -> None: + cli, log, runtime, registry, state = fixture + activate(cli) + guard = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-blocked") + packet = guard["replan_action_packet"] + task = cli("todo", "add", "--goal-id", GOAL, "--role", "agent", "--claimed-by", AGENT, + "--task-class", "advancement_task", "--action-kind", "joint_probe", "--target-key", "joint", + "--explore-result-node-ref", "joint", "--replan-obligation-id", packet["obligation_id"], + "--text", "Run the bounded joint experiment.") + blocker = cli("todo", "add", "--goal-id", GOAL, "--role", "agent", "--claimed-by", AGENT, + "--task-class", "blocker", "--action-kind", "investigate", "--unblocks-todo-id", task["todo_id"], + "--text", "Resolve the bounded dependency.") + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="blocked", blocked_reason="An identified dependency remains open.")) + result = observation("joint") + result["progress"].update(work_item_id=task["todo_id"], result_class="blocked", blocker_id=blocker["todo_id"]) + result["progress"].pop("coverage_complete") + result["closure_basis"] = None + result["input_observations"] = guard["bounded_research_frontier"]["selected_gap"]["input_observations"] + result["execution_lineage"] = {"schema_version": "research_execution_lineage_v0", "goal_id": GOAL, + "gap_id": packet["capability_guard"]["gap_id"], "replan_obligation_id": packet["obligation_id"], + "successor_todo_id": task["todo_id"], "agent_id": AGENT} + result["composition_resolution"] = {"schema_version": "research_composition_resolution_v0", + "disposition": "deferred", "evidence_ids": ["ev-joint"]} + recorded = append_research_observation(log, goal_id=GOAL, observation=result, agent_id=AGENT, + registry_path=registry, runtime_root=runtime) + cli("todo", "update", "--goal-id", GOAL, "--todo-id", task["todo_id"], "--agent-id", AGENT, + "--status", "blocked", "--resume-when", f"todo_done:{blocker['todo_id']}", "--reason", "Await the bounded dependency.") + frontier = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT)["research_execution_frontier"] + assert frontier["deferred_count"] == 1 and frontier["observed_count"] == 0 + refresh = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", packet["obligation_id"], "--turn-instance-id", "fixture-blocked", + "--classification", "bounded_fixture_blocker", "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_progress", + "--progress-work-item-id", task["todo_id"], "--progress-result-class", "blocked", + "--progress-blocker-id", blocker["todo_id"], "--progress-coverage-scope-id", "scope-joint", "--progress-evidence-id", "ev-joint", + "--vision-summary", "Test this explicit research question.", "--vision-acceptance", "The typed experiment addresses the question.", + "--vision-replan-trigger", "The experiment remains blocked with a typed resume condition.") + assert refresh["ok"] + saved = json.loads(Path(refresh["json_path"]).read_text()) + assert saved["autonomous_replan_ack"]["semantic_delta"]["capability_outcome"] == "composition_temporarily_deferred" + assert saved["progress_observation"] == recorded["observation"]["progress"] + spent = cli("quota", "spend-slot", "--goal-id", GOAL, "--agent-id", AGENT, "--slots", "1", "--source", "heartbeat", + "--execute", "--replan-obligation-id", packet["obligation_id"], "--turn-instance-id", "fixture-blocked") + assert spent["settlement_progress"]["state"] == "settled" + source = json.loads(registry.read_text())["goals"][0] + _, repeated = qualify_replan_writeback(newest_first_runs=[{"agent_id": AGENT, + "progress_observation": recorded["observation"]["progress"]}], state_text=state.read_text(), agent_id=AGENT, + goal_id=GOAL, registry_goal=source, guard_scoped=True, + capability_evidence=prepare_research_replan_evidence(runtime_root=runtime, goal_id=GOAL, + agent_id=AGENT, registry_goal=source, state_text=state.read_text(), capability_guard=packet["capability_guard"]), + guard_semantic_replan_obligation_id=packet["obligation_id"], guard_capability=packet["capability_guard"], + progress_observation=recorded["observation"]["progress"]) + assert repeated["accepted"] is False + cli("todo", "complete", "--goal-id", GOAL, "--todo-id", blocker["todo_id"], "--agent-id", AGENT, + "--no-follow-up", "--evidence", "Dependency resolved in the fixture.") + frontier = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT)["research_execution_frontier"] + assert frontier["deferred_count"] == 0 and frontier["pending_count"] == 1 + cli("todo", "update", "--goal-id", GOAL, "--todo-id", task["todo_id"], "--agent-id", AGENT, + "--status", "open", "--clear-resume-when", "--reason", "Dependency resolved; resume the experiment.") + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="open")) + frontier = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT)["research_execution_frontier"] + assert frontier["scheduled_count"] == 1 and frontier["observed_count"] == 0 + + +@pytest.mark.parametrize("disposition", ["observed", "dismissed"]) +def test_real_replan_guard_accepts_exact_result_source_and_candidate_dismissal(fixture, disposition: str) -> None: + cli, log, runtime, registry, _state = fixture + activate(cli) + guard = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-terminal") + packet = guard["replan_action_packet"] + task = cli("todo", "add", "--goal-id", GOAL, "--role", "agent", "--claimed-by", AGENT, + "--task-class", "advancement_task", "--action-kind", "joint_probe", "--target-key", "joint", + "--explore-result-node-ref", "joint", "--replan-obligation-id", packet["obligation_id"], + "--text", "Run the bounded joint experiment.") + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="dead_end" if disposition == "dismissed" else "resolved")) + result = observation("joint") + result["progress"]["work_item_id"] = task["todo_id"] + result["input_observations"] = guard["bounded_research_frontier"]["selected_gap"]["input_observations"] + result["execution_lineage"] = {"schema_version": "research_execution_lineage_v0", "goal_id": GOAL, + "gap_id": packet["capability_guard"]["gap_id"], "replan_obligation_id": packet["obligation_id"], + "successor_todo_id": task["todo_id"], "agent_id": AGENT} + if disposition == "dismissed": + result["progress"]["result_class"] = "no_followup" + result["closure_basis"]["disposition"] = "no_followup" + result["composition_resolution"] = {"schema_version": "research_composition_resolution_v0", + "disposition": "dismissed", "basis": "outside_scope", "evidence_ids": ["ev-joint"]} + recorded = append_research_observation(log, goal_id=GOAL, observation=result, agent_id=AGENT, + registry_path=registry, runtime_root=runtime) + refresh = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", packet["obligation_id"], "--turn-instance-id", "fixture-terminal", + "--classification", "bounded_fixture_terminal", "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_progress", + "--progress-work-item-id", task["todo_id"], "--progress-result-class", result["progress"]["result_class"], + "--progress-coverage-scope-id", "scope-joint", "--progress-coverage-complete", "--progress-evidence-id", "ev-joint", + "--vision-summary", "Test this explicit research question.", "--vision-acceptance", "The typed result addresses the question.", + "--vision-replan-trigger", "Retain the remaining Goal acceptance beyond this bounded candidate.") + saved = json.loads(Path(refresh["json_path"]).read_text()) + expected = "composition_candidate_dismissed" if disposition == "dismissed" else "composition_experiment_observed" + assert saved["autonomous_replan_ack"]["semantic_delta"]["capability_outcome"] == expected + assert saved["progress_observation"] == recorded["observation"]["progress"] + spent = cli("quota", "spend-slot", "--goal-id", GOAL, "--agent-id", AGENT, "--slots", "1", "--source", "heartbeat", + "--execute", "--replan-obligation-id", packet["obligation_id"], "--turn-instance-id", "fixture-terminal") + assert spent["settlement_progress"]["state"] == "settled" + complete = cli("todo", "supersede" if disposition == "dismissed" else "complete", "--goal-id", GOAL, + "--todo-id", task["todo_id"], "--agent-id", AGENT, + *( ["--reason", "Candidate dismissal is recorded."] if disposition == "dismissed" + else ["--evidence", "Typed scoped evidence is recorded.", "--no-follow-up"])) + assert complete["capability_completion_evidence"]["disposition"] == ( + "candidate_dismissed" if disposition == "dismissed" else "experiment_observed") + + +@pytest.mark.parametrize("change", ["input", "scope", "disabled"]) +def test_invalidated_original_turn_retires_with_no_spend_and_current_frontier_is_preserved(fixture, change: str) -> None: + cli, log, _runtime, _registry, state = fixture + activate(cli) + original = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--codex-app", + "--turn-instance-id", "fixture-invalidated") + packet = original["replan_action_packet"] + if change == "input": + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="a", title="Research a", status="open")) + elif change == "scope": + cli("configure-goal", "--goal-id", GOAL, "--explore-composition-scope-id", "replacement-scope", "--execute") + else: + cli("configure-goal", "--goal-id", GOAL, "--explore-composition-mode", "disabled", "--execute") + identity = ["--goal-id", GOAL, "--agent-id", AGENT, "--replan-obligation-id", packet["obligation_id"], + "--turn-instance-id", "fixture-invalidated"] + rejected = cli("refresh-state", *identity, "--classification", "fixture_retirement_probe", + "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_progress", + "--progress-result-class", "advanced", "--progress-evidence-id", "unrelated", success=False) + contract = rejected["replan_transition"]["retirement_contract"] + assert contract["blocking_todo_count"] == 0 + assert contract["original_guard"] == packet["capability_guard"] + progress = contract["progress_observation"] + # This is a causal lifecycle exit, not a way to count progress or debit. + invalid_progress = cli("refresh-state", *identity, "--classification", "fixture_false_progress", + "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_progress", + "--progress-result-class", "blocked", "--progress-blocker-id", progress["blocker_id"], + "--progress-evidence-id", progress["evidence_ids"][0], success=False) + assert "requires outcome_gap" in json.dumps(invalid_progress) + retired = cli("refresh-state", *identity, "--classification", "fixture_duty_invalidated", + "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_gap", + "--progress-result-class", "blocked", "--progress-blocker-id", progress["blocker_id"], + "--progress-evidence-id", progress["evidence_ids"][0], + "--vision-summary", "Continue the bounded research question from current evidence.", + "--vision-acceptance", "Authoritative evidence must satisfy the research question.", + "--vision-replan-trigger", "Reassess the current frontier after this admitted basis changed.") + assert retired["settlement_progress"]["state"] == "settled" + assert retired["settlement_progress"]["closeout_kind"] == "capability_duty_retired_no_spend" + assert "settlement_owed" not in retired + denied = cli("quota", "spend-slot", *identity, "--slots", "1", "--source", "heartbeat", "--execute") + assert denied["appended"] is False and denied["idempotent_replay"] is True + assert denied["settlement_progress"]["closeout_kind"] == "capability_duty_retired_no_spend" + assert not any(receipt["step_kind"] == "quota_spend" for receipt in denied["settlement_result"]["receipts"]) + replay = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, + "--turn-instance-id", "fixture-invalidated", "--codex-app") + assert replay["heartbeat_receipt"]["semantic_replan_capability_guard"] == packet["capability_guard"] + assert replay["effective_action"] == "heartbeat_settled_skip" + fresh = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--codex-app", + "--turn-instance-id", "fixture-after-retirement") + assert fresh["effective_action"] != "unsettled_host_turn_recovery" + assert fresh["goal_frontier_projection"]["acceptance_gaps"] + assert "quota_slot_spent" not in json.dumps(retired) + assert "completed_at=" not in state.read_text() + + +def test_retirement_cannot_hide_a_runnable_bound_task_and_keeps_that_task_open(fixture) -> None: + cli, _log, _runtime, _registry, _state = fixture + activate(cli) + original = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--codex-app", + "--turn-instance-id", "fixture-active-duty") + packet = original["replan_action_packet"] + task = cli("todo", "add", "--goal-id", GOAL, "--role", "agent", "--claimed-by", AGENT, + "--task-class", "advancement_task", "--action-kind", "joint_probe", "--target-key", "joint", + "--explore-result-node-ref", "joint", "--replan-obligation-id", packet["obligation_id"], "--text", "Run the bounded experiment.") + cli("configure-goal", "--goal-id", GOAL, "--explore-composition-scope-id", "replacement-scope", "--execute") + base = ["refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, "--replan-obligation-id", packet["obligation_id"], + "--turn-instance-id", "fixture-active-duty", "--classification", "fixture_retirement", "--delivery-batch-scale", "implementation", + "--delivery-outcome", "outcome_gap", "--progress-result-class", "blocked"] + rejected = cli(*base, "--progress-blocker-id", "unrelated", "--progress-evidence-id", "unrelated", success=False) + contract = rejected["replan_transition"]["retirement_contract"] + assert contract["blocking_todo_ids"] == [task["todo_id"]] + progress = contract["progress_observation"] + cli(*base, "--progress-blocker-id", progress["blocker_id"], "--progress-evidence-id", progress["evidence_ids"][0], success=False) + assert cli("todo", "list", "--goal-id", GOAL, "--todo-id", task["todo_id"])["todos"][0]["status"] == "open" + # An explicit lifecycle pause keeps work and evidence visible; retirement + # still closes only the original duty, never the Todo or its Goal. + cli("todo", "update", "--goal-id", GOAL, "--todo-id", task["todo_id"], "--agent-id", AGENT, + "--status", "blocked", "--clear-resume-when", "--reason", "The admitted basis changed; reassess before execution.") + retired = cli(*base, "--progress-blocker-id", progress["blocker_id"], "--progress-evidence-id", progress["evidence_ids"][0], + "--vision-summary", "Reassess the bounded research question.", "--vision-acceptance", "Current evidence must satisfy the question.", + "--vision-replan-trigger", "Keep the acceptance and paused work visible.") + assert retired["settlement_progress"]["closeout_kind"] == "capability_duty_retired_no_spend" + assert cli("todo", "list", "--goal-id", GOAL, "--todo-id", task["todo_id"])["todos"][0]["status"] == "blocked" diff --git a/tests/control_plane/test_declared_terminal_settlement_cli.py b/tests/control_plane/test_declared_terminal_settlement_cli.py index 9b0f33a92b..45492ba092 100644 --- a/tests/control_plane/test_declared_terminal_settlement_cli.py +++ b/tests/control_plane/test_declared_terminal_settlement_cli.py @@ -12,11 +12,26 @@ import test_quota_settlement_cli as cli from canonical_authority_fixture import initialize_canonical_authority, isolate_sqlite_runtime +from loopx.cli_commands.todo import _completion_hook_state_version from loopx.control_plane.coordination.runtime_shadow import build_todo_runtime_shadow_projection from loopx.control_plane.quota.settlement import read_heartbeat_settlement from loopx.control_plane.todos.markdown import render_todo_markdown +def test_same_second_todo_closeout_has_distinct_replay_stable_hook_version(): + committed_at = "2026-09-30T13:33:32-07:00" + ordinary = _completion_hook_state_version( + {"completion_continuation": "active_goal"}, committed_at, + ) + terminal = _completion_hook_state_version( + {"completion_continuation": "no_followup"}, committed_at, + ) + assert ordinary != terminal + assert terminal == _completion_hook_state_version( + {"completion_continuation": "no_followup"}, committed_at, + ) + + def _fixture(tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str): # Scope managed Effect discovery as well as provider data for both arms. isolate_sqlite_runtime(tmp_path, monkeypatch) diff --git a/tests/control_plane/test_goal_frontier_replan_rules.py b/tests/control_plane/test_goal_frontier_replan_rules.py index 18b04d3d3f..65a32cca1c 100644 --- a/tests/control_plane/test_goal_frontier_replan_rules.py +++ b/tests/control_plane/test_goal_frontier_replan_rules.py @@ -159,6 +159,27 @@ def test_goal_frontier_replan_decision_table( ) +def test_capability_evidence_gap_precedence_and_disabled_parity() -> None: + # A caller-owned gap enters before monitor fallback. Existing authority, + # runnable work and acceptance checkpoints retain their established order. + selected = select_goal_frontier_replan_rule(GoalFrontierReplanFacts(capability_gap_pending=True)) + assert selected.rule is GoalFrontierReplanRule.CAPABILITY_EVIDENCE_GAP + assert selected.derives_obligation + for fields, expected in [ + ({"existing_replan_required": True}, GoalFrontierReplanRule.EXISTING_OBLIGATION), + ({"blocking_handoff_gate_count": 1}, GoalFrontierReplanRule.BLOCKING_HANDOFF_GATE), + ({"ready_deferred_successor_count": 1}, GoalFrontierReplanRule.READY_DEFERRED_SUCCESSOR), + ({"blocking_user_open_count": 1}, GoalFrontierReplanRule.OPEN_USER_TODO), + ({"acceptance_gap_count": 1}, GoalFrontierReplanRule.VISION_ACCEPTANCE_GAP), + ({"long_todo_chain_triggered": True}, GoalFrontierReplanRule.LONG_TODO_CHAIN), + ({"current_agent_blocker_count": 1}, GoalFrontierReplanRule.CURRENT_AGENT_BLOCKER), + ({"selectable_frontier_advancement": 1}, GoalFrontierReplanRule.NOT_MONITOR_ONLY), + ]: + decision = select_goal_frontier_replan_rule(GoalFrontierReplanFacts(capability_gap_pending=True, **fields)) + assert decision.rule is expected + assert select_goal_frontier_replan_rule(GoalFrontierReplanFacts()).rule is GoalFrontierReplanRule.NOT_MONITOR_ONLY + + def _repeat_vision_gap() -> list[dict[str, object]]: return [ { diff --git a/tests/control_plane/test_prompt_upgrade_hook.py b/tests/control_plane/test_prompt_upgrade_hook.py index 3317889115..115a61994e 100644 --- a/tests/control_plane/test_prompt_upgrade_hook.py +++ b/tests/control_plane/test_prompt_upgrade_hook.py @@ -158,6 +158,13 @@ def test_upgrade_read_projection_preserves_work_authority(tmp_path, monkeypatch, receipt.write_bytes(contents) pending = build_live_quota_should_run_decision(status, **kwargs) assert pending["required_reads"][-1]["kind"] == "automation_prompt_upgrade" + assert "turn_start_capability_hook_dispatch" not in baseline + dispatch = pending["turn_start_capability_hook_dispatch"] + assert set(dispatch) == {"required_reads"} + assert len(dispatch["required_reads"]) == 1 + assert dispatch["required_reads"][0]["kind"] == pending["required_reads"][-1]["kind"] + assert dispatch["required_reads"][0]["command"] == pending["required_reads"][-1]["command"] + assert pending["required_reads"][-1]["source"] == "turn_start_capability_hook" assert pending["interaction_contract"]["agent_channel"]["required_reads"] == pending["required_reads"] hint = pending["required_reads"][-1] assert len(hint["command"]) > 360 diff --git a/tests/control_plane/test_research_execution_authority.py b/tests/control_plane/test_research_execution_authority.py new file mode 100644 index 0000000000..409afe327b --- /dev/null +++ b/tests/control_plane/test_research_execution_authority.py @@ -0,0 +1,323 @@ +"""Execution evidence uses the real Todo provider, never a stale display row.""" +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + +import pytest +from canonical_authority_fixture import initialize_canonical_authority, isolate_sqlite_runtime + +from loopx.capabilities.explore.research_evidence import append_research_observation +from loopx.capabilities.explore.result_log import ( + append_explore_result_event, build_explore_edge_event, build_explore_node_event, + build_explore_result_projection, explore_result_log_path, load_explore_result_events_strict, +) +from loopx.control_plane.coordination.runtime_shadow import build_todo_runtime_shadow_projection +from loopx.control_plane.effect_runtime import restart_effect_runtime +from loopx.todos import complete_goal_todo +from loopx.capabilities.explore.research_frontier import build_research_composition_frontier, prepare_research_replan_evidence + + +def observation(node: str) -> dict: + return { + "schema_version": "typed_research_observation_v0", "explore_node_id": node, + "progress": {"schema_version": "typed_progress_observation_v0", "work_item_id": f"todo_{node}", + "result_class": "exploration_exhausted", "coverage_scope_id": f"scope-{node}", + "coverage_complete": True, "evidence_ids": [f"ev-{node}"]}, + "closure_basis": {"schema_version": "research_closure_basis_v0", "disposition": "bounded", + "constraints": [{"kind": "invariant", "id": "boundary", "role": "decisive"}], + "evidence_ids": [f"ev-{node}"]}, + } + + +@pytest.mark.parametrize("provider", ["file", "sqlite"]) +@pytest.mark.parametrize("canonical_owner", ["fixture-agent", "other-agent"]) +def test_execution_attribution_reads_promoted_provider( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, canonical_owner: str, +) -> None: + isolate_sqlite_runtime(tmp_path, monkeypatch) + try: + goal = "research-provider-fixture" + state = tmp_path / "ACTIVE_GOAL_STATE.md" + state.write_text("# Goal\n\n## Agent Todo\n\n") + runtime = tmp_path / "runtime" + registry = tmp_path / "registry.json" + registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{ + "id": goal, "repo": str(tmp_path), "state_file": state.name, "status": "active", + "coordination": {"agent_model": "peer_v1", "registered_agents": ["fixture-agent", "other-agent"]}, + }]})) + task = { + "schema_version": "todo_item_v0", "todo_id": "todo_joint", "index": 1, + "role": "agent", "status": "open", "done": False, "text": "Run the bounded joint experiment.", + "archive_state": "active", "source_section": "Agent Todo", "priority": "P1", + "claimed_by": canonical_owner, "task_class": "advancement_task", "action_kind": "joint_probe", + "target_key": "joint", "explore_result_node_refs": ["joint"], + "replan_obligation_id": "replan-0123456789abcdef", + } + initialize_canonical_authority(runtime, goal, + build_todo_runtime_shadow_projection(goal_id=goal, todos=[task]), state_path=state, provider=provider) + # Even a plausible display claim is not the promoted source of truth. + state.write_text("# Goal\n\n## Agent Todo\n\n- [ ] Stale display\n" + " \n") + log = explore_result_log_path(runtime, goal) + for name in ["a", "b", "joint"]: + append_explore_result_event(log, build_explore_node_event( + goal_id=goal, node_id=name, title=f"Research {name}", status="resolved", + node_kind="experiment" if name == "joint" else "hypothesis")) + append_research_observation(log, goal_id=goal, observation=observation("b")) + source = observation("a") + source["composition_candidates"] = [{"basis": "explicit", "target_node_id": "b", + "interaction_kind": "state_interference", "evidence_ids": ["ev-a", "ev-b"]}] + append_research_observation(log, goal_id=goal, observation=source) + for name in ["a", "b"]: + append_explore_result_event(log, build_explore_edge_event( + goal_id=goal, from_node="joint", to_node=name, edge_type="depends_on")) + view = build_explore_result_projection(load_explore_result_events_strict(log, goal_id=goal), goal_id=goal) + gap = view["research_frontier"]["gaps"][0] + result = observation("joint") + result["input_observations"] = gap["input_observations"] + result["execution_lineage"] = { + "schema_version": "research_execution_lineage_v0", "goal_id": goal, "gap_id": gap["gap_id"], + "replan_obligation_id": task["replan_obligation_id"], "successor_todo_id": "todo_joint", "agent_id": "fixture-agent", + } + before = log.read_bytes(), state.read_bytes() + if canonical_owner == "fixture-agent": + receipt = append_research_observation(log, goal_id=goal, observation=result, agent_id="fixture-agent", + registry_path=registry, runtime_root=runtime) + assert receipt["written"] + else: + with pytest.raises(ValueError, match="same-agent runnable"): + append_research_observation(log, goal_id=goal, observation=result, agent_id="fixture-agent", + registry_path=registry, runtime_root=runtime) + assert log.read_bytes() == before[0] + assert state.read_bytes() == before[1] + finally: + restart_effect_runtime() + + +@pytest.mark.parametrize("provider", ["file", "sqlite"]) +@pytest.mark.parametrize("resolution", ["observed", "dismissed", "deferred"]) +@pytest.mark.parametrize("diagnostic_timing", [None, "before_activation", "after_activation"]) +def test_native_completion_refuses_missing_result_and_accepts_exact_observation( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, resolution: str, diagnostic_timing: str | None, +) -> None: + isolate_sqlite_runtime(tmp_path, monkeypatch) + try: + goal, agent = "research-native-fixture", "fixture-agent" + state = tmp_path / "ACTIVE_GOAL_STATE.md" + state.write_text("# Goal\n\n## Agent Todo\n\n") + runtime, registry = tmp_path / "runtime", tmp_path / "registry.json" + harness = {"enabled": True, "composition_mode": "explicit_only", "composition_scope_id": "scope-joint"} + registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{ + "id": goal, "repo": str(tmp_path), "state_file": state.name, "status": "active", + "spawn_policy": {"explore_harness": {"enabled": True} if diagnostic_timing == "before_activation" else harness}, + "coordination": {"agent_model": "peer_v1", "registered_agents": [agent]}, + }]})) + def cli(*args: str, success: bool = True) -> dict: + process = subprocess.run([sys.executable, "-m", "loopx.entrypoint", "--format", "json", + "--registry", str(registry), "--runtime-root", str(runtime), *args], + cwd=Path(__file__).resolve().parents[2], capture_output=True, text=True, timeout=60) + assert (process.returncode == 0) is success, process.stdout + process.stderr + return json.loads(process.stdout) + + packet = tmp_path / "observation.json" + def record(value: dict, *, success: bool = True) -> dict: + packet.write_text(json.dumps(value)) + return cli("explore", "observe", "--goal-id", goal, "--agent-id", agent, + "--observation-json", str(packet), success=success) + + log = explore_result_log_path(runtime, goal) + for name in ["a", "b", "joint"]: + append_explore_result_event(log, build_explore_node_event( + goal_id=goal, node_id=name, title=f"Research {name}", + status="open" if name == "joint" else "resolved", + node_kind="experiment" if name == "joint" else "hypothesis")) + append_research_observation(log, goal_id=goal, observation=observation("b")) + source = observation("a") + source["composition_candidates"] = [{"basis": "explicit", "target_node_id": "b", + "interaction_kind": "state_interference", "evidence_ids": ["ev-a", "ev-b"]}] + append_research_observation(log, goal_id=goal, observation=source) + for name in ["a", "b"]: + append_explore_result_event(log, build_explore_edge_event( + goal_id=goal, from_node="joint", to_node=name, edge_type="depends_on")) + if diagnostic_timing: + append_explore_result_event(log, build_explore_node_event(goal_id=goal, node_id="diagnostic", + title="Retained diagnostic", node_kind="experiment", status="resolved")) + for name in ["a", "b"]: + append_explore_result_event(log, build_explore_edge_event(goal_id=goal, + from_node="diagnostic", to_node=name, edge_type="depends_on")) + cold = build_explore_result_projection(load_explore_result_events_strict(log, goal_id=goal), goal_id=goal) + diagnostic = observation("diagnostic") + diagnostic["input_observations"] = cold["research_frontier"]["gaps"][0]["input_observations"] + assert record(diagnostic)["written"] + if diagnostic_timing == "before_activation": + assert cli("configure-goal", "--goal-id", goal, "--explore-composition-mode", "explicit_only", + "--explore-composition-scope-id", "scope-joint", "--execute")["ok"] + events = load_explore_result_events_strict(log, goal_id=goal) + projection = build_explore_result_projection(events, goal_id=goal) + frontier = build_research_composition_frontier(projection, + candidate_sources=[{"node_id": e["result_id"], "research_observation": e["research_observation"]} + for e in events if e.get("research_observation")], + harness=harness, todos=[], agent_id=agent) + gap = frontier["selected_gap"] + assert gap["status"] == "pending" + if diagnostic_timing: + assert projection["research_frontier"]["observed_count"] == 1 + assert frontier["observed_count"] == 0 + task = {"schema_version": "todo_item_v0", "todo_id": "todo_joint", "index": 1, + "role": "agent", "status": "open", "done": False, "text": "Run the bounded joint experiment.", + "archive_state": "active", "source_section": "Agent Todo", "priority": "P1", + "claimed_by": agent, "task_class": "advancement_task", "action_kind": "joint_probe", + "target_key": "joint", "explore_result_node_refs": ["joint"], "replan_obligation_id": gap["obligation_id"]} + initialize_canonical_authority(runtime, goal, + build_todo_runtime_shadow_projection(goal_id=goal, todos=[task]), state_path=state, provider=provider) + live = cli("explore", "summary", "--goal-id", goal, "--agent-id", agent)["research_execution_frontier"] + assert live["scheduled_count"] == 1 and live["observed_count"] == 0 + from loopx.control_plane.work_items.task_lease import acquire_task_lease + acquired = acquire_task_lease(registry_path=registry, runtime_root=runtime, goal_id=goal, + todo_id="todo_joint", owner=agent, idempotency_key="native-research-proof", ttl_seconds=300) + assert acquired["ok"] + from loopx.control_plane.coordination.local_authority import read_canonical_todos_if_promoted, LocalCoordinationAuthorityUnavailable + from loopx.control_plane.todos import provider_terminal_lifecycle as terminal_adapter + requests = [] + native_call = terminal_adapter.effect_runtime_result + def capture_native(method, params, **kwargs): + if method == "coordination.local_authority.todo_terminal": + requests.append(json.loads(json.dumps(params))) + return native_call(method, params, **kwargs) + monkeypatch.setattr(terminal_adapter, "effect_runtime_result", capture_native) + before = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + with pytest.raises(ValueError, match="current typed experiment observation"): + complete_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", + agent_id=agent, claimed_by=agent, no_followup=True, evidence="Synthetic result required.", + task_lease_idempotency_key="native-research-proof", task_lease_expected_version=acquired["lease"]["version"]) + forged = {**requests[-1], "operation_identity": {"kind": "explicit", "operation_id": "native-forged-approval"}, + "capability_completion_evidence": {"approved": True}} + direct = native_call("coordination.local_authority.todo_terminal", forged) + assert direct["changed"] is False + assert direct["reason_code"] == "research_experiment_result_required" + retired = native_call("coordination.local_authority.todo_terminal", {**forged, "command": "supersede", + "operation_identity": {"kind": "explicit", "operation_id": "native-supersede-without-result"}, + "requested_no_followup": False, "reason": "Attempt to close through another verb.", "completion_policy_request": None}) + assert retired["changed"] is False + assert retired["reason_code"] == "research_experiment_result_required" + after = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + assert after["provider_revision"] == before["provider_revision"] + assert after["todos"][0]["status"] == "open" + append_explore_result_event(log, build_explore_node_event(goal_id=goal, node_id="joint", + title="Research joint", node_kind="experiment", + status={"observed": "resolved", "dismissed": "dead_end", "deferred": "blocked"}[resolution], + blocked_reason="An identified dependency remains open." if resolution == "deferred" else None)) + result = observation("joint") + result["input_observations"] = gap["input_observations"] + result["execution_lineage"] = {"schema_version": "research_execution_lineage_v0", "goal_id": goal, + "gap_id": gap["gap_id"], "replan_obligation_id": gap["obligation_id"], "successor_todo_id": "todo_joint", "agent_id": agent} + if resolution == "dismissed": + result["progress"]["result_class"] = "no_followup" + result["closure_basis"]["disposition"] = "no_followup" + result["composition_resolution"] = {"schema_version": "research_composition_resolution_v0", + "disposition": "dismissed", "basis": "outside_scope", "evidence_ids": ["ev-joint"]} + elif resolution == "deferred": + from loopx.todos import add_goal_todo + blocker = add_goal_todo(registry_path=registry, goal_id=goal, role="agent", + text="Resolve the bounded dependency.", task_class="blocker", action_kind="investigate", + claimed_by=agent, agent_id=agent, unblocks_todo_id="todo_joint") + result["progress"].update(result_class="blocked", coverage_complete=False, blocker_id=blocker["todo_id"]) + result["closure_basis"] = None + result["composition_resolution"] = {"schema_version": "research_composition_resolution_v0", + "disposition": "deferred", "evidence_ids": ["ev-joint"]} + recorded = record(result) + written = log.read_bytes() + replay = record(result) + assert replay["replayed"] and not replay["written"] + assert log.read_bytes() == written + if resolution != "deferred": + changed = json.loads(json.dumps(result)) + changed["progress"]["evidence_ids"].append("ev-new") + assert "pending gap" in json.dumps(record(changed, success=False)) + assert log.read_bytes() == written + if resolution == "deferred": + from loopx.todos import update_goal_todo + from loopx.capabilities.explore.composition_frontier import project_live_explore_composition_frontier + from loopx.control_plane.work_items.semantic_replan_writeback import qualify_replan_writeback + from loopx.control_plane.work_items.task_lease import release_task_lease + with pytest.raises(LocalCoordinationAuthorityUnavailable, match="Release the active execution lease"): + update_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", agent_id=agent, + status="blocked", resume_when=f"todo_done:{blocker['todo_id']}", reason="Await the bounded dependency.") + released = release_task_lease(registry_path=registry, runtime_root=runtime, goal_id=goal, + todo_id="todo_joint", owner=agent, idempotency_key="native-research-proof", + expected_version=acquired["lease"]["version"]) + assert released["ok"] + update_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", agent_id=agent, + status="blocked", resume_when=f"todo_done:{blocker['todo_id']}", reason="Await the bounded dependency.") + source = json.loads(registry.read_text())["goals"][0] + def live_deferred(): + return project_live_explore_composition_frontier(runtime_root=runtime, goal_id=goal, + agent_id=agent, status_payload={"run_history": {"goals": [source]}}) + assert live_deferred()["deferred_count"] == 1 + _, delta = qualify_replan_writeback(newest_first_runs=[], state_text=state.read_text(), agent_id=agent, + goal_id=goal, registry_goal=source, guard_scoped=True, + capability_evidence=prepare_research_replan_evidence(runtime_root=runtime, goal_id=goal, + agent_id=agent, registry_goal=source, state_text=state.read_text()), + guard_semantic_replan_obligation_id=gap["obligation_id"], progress_observation=recorded["observation"]["progress"], + guard_capability={"capability_id": "explore", "gap_id": gap["gap_id"], "frontier_revision": gap["frontier_revision"]}) + assert delta["capability_outcome"] == "composition_temporarily_deferred" + revision = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal)["provider_revision"] + with pytest.raises(ValueError): + complete_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", agent_id=agent, + no_followup=True, task_lease_idempotency_key="native-research-proof", + task_lease_expected_version=acquired["lease"]["version"]) + assert read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal)["provider_revision"] == revision + dependency_lease = acquire_task_lease(registry_path=registry, runtime_root=runtime, goal_id=goal, + todo_id=blocker["todo_id"], owner=agent, idempotency_key="dependency-resolved", ttl_seconds=300) + complete_goal_todo(registry_path=registry, goal_id=goal, todo_id=blocker["todo_id"], agent_id=agent, + no_followup=True, evidence="Dependency resolved in the synthetic fixture.", + task_lease_idempotency_key="dependency-resolved", task_lease_expected_version=dependency_lease["lease"]["version"]) + assert live_deferred()["deferred_count"] == 0 + assert live_deferred()["pending_count"] == 1 + update_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", agent_id=agent, + status="open", clear_resume_when=True, reason="Dependency resolved; resume this experiment.") + append_explore_result_event(log, build_explore_node_event(goal_id=goal, node_id="joint", + title="Research joint", node_kind="experiment", status="open")) + assert live_deferred()["scheduled_count"] == 1 + assert live_deferred()["observed_count"] == 0 + reacquired = acquire_task_lease(registry_path=registry, runtime_root=runtime, goal_id=goal, + todo_id="todo_joint", owner=agent, idempotency_key="native-research-resumed", ttl_seconds=300) + assert reacquired["ok"] + assert reacquired["lease"]["version"] > acquired["lease"]["version"] + return + accepted = complete_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", + agent_id=agent, claimed_by=agent, clear_claim=True, no_followup=True, evidence="Typed synthetic result recorded.", + task_lease_idempotency_key="native-research-proof", task_lease_expected_version=acquired["lease"]["version"]) + assert accepted["completed"] + assert accepted["capability_completion_evidence"]["experiment_node_id"] == "joint" + completed = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + assert completed["todos"][0]["status"] == "done" + assert completed["todos"][0].get("claimed_by") is None + from loopx.control_plane.todos.provider_terminal_lifecycle import archive_canonical_todos_if_promoted + archived = archive_canonical_todos_if_promoted(registry_path=registry, runtime_root=runtime, + goal_id=goal, role="agent", max_active_done=0, dry_run=False) + assert archived["moved_todo_ids"] == ["todo_joint"] + completed = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + assert completed["todos"][0]["archive_state"] == "archive" + from loopx.capabilities.explore.composition_frontier import project_live_explore_composition_frontier + def live(): + return project_live_explore_composition_frontier(runtime_root=runtime, goal_id=goal, + agent_id=agent, status_payload={"run_history": {"goals": json.loads(registry.read_text())["goals"]}}) + assert live()[f"{resolution}_count"] == 1 + assert live()["scheduled_count"] == 0 + cli_view = cli("explore", "summary", "--goal-id", goal, "--agent-id", agent)["research_execution_frontier"] + assert cli_view[f"{resolution}_count"] == 1 and cli_view["scheduled_count"] == 0 + append_explore_result_event(log, build_explore_node_event(goal_id=goal, node_id="a", title="Research a", status="open")) + assert live()["observed_count"] == 0 + replay = complete_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", + agent_id=agent, claimed_by=agent, clear_claim=True, no_followup=True, evidence="Historical typed result remains immutable.", + task_lease_idempotency_key="native-research-proof", task_lease_expected_version=acquired["lease"]["version"]) + assert replay["provider_status"] == "replayed" + assert replay["capability_completion_evidence"] == accepted["capability_completion_evidence"] + latest = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + assert latest["provider_revision"] == completed["provider_revision"] + finally: + restart_effect_runtime() diff --git a/tests/control_plane/test_runtime_source_read_batching.py b/tests/control_plane/test_runtime_source_read_batching.py index 1cd1f167c6..61122c68f2 100644 --- a/tests/control_plane/test_runtime_source_read_batching.py +++ b/tests/control_plane/test_runtime_source_read_batching.py @@ -1,6 +1,7 @@ from __future__ import annotations import hashlib +import sys from pathlib import Path import pytest @@ -9,8 +10,10 @@ def serial_fingerprint(root: Path) -> str: - """Independent reference: names and raw bytes, not decoded source text.""" + """Independent reference: selected Python adapter, names, and raw bytes.""" digest = hashlib.sha256() + digest.update(sys.executable.encode("utf-8")) + digest.update(str(sys.version_info[:3]).encode("ascii")) for path in sorted(p for p in root.rglob("*") if p.suffix in {".ts", ".json"}): digest.update(path.relative_to(root).as_posix().encode("utf-8")) digest.update(path.read_bytes()) diff --git a/tests/control_plane_ts/blocked_hard_lease_lifecycle.test.ts b/tests/control_plane_ts/blocked_hard_lease_lifecycle.test.ts index 3ae99b0989..4244b47134 100644 --- a/tests/control_plane_ts/blocked_hard_lease_lifecycle.test.ts +++ b/tests/control_plane_ts/blocked_hard_lease_lifecycle.test.ts @@ -123,6 +123,27 @@ for (const provider of ["file", "sqlite"] as const) { assert.equal((after.head.leases as JsonObject[])[0]!.status, "released"); }); + providerTest(`${provider}: typed prerequisite wait preserves pause fencing and grants no execution`, async () => { + const store = await seeded(provider, "2026-09-25T05:00:00Z"); + const input = {...pause("pause-typed-wait"), planning_intent: { + status: "blocked", reason: "Await the identified prerequisite", resume_when: "todo_done:todo_successor"}}; + assert.equal((await executeCoordinationTodoUpdate(store, input)).status, "applied"); + const after = await read(store), todo = (after.head.todos as JsonObject[])[0]!; + assert.equal(todo.status, "blocked"); + assert.equal(todo.resume_when, "todo_done:todo_successor"); + assert.equal(todo.claimed_by, OWNER); + assert.equal((after.head.leases as JsonObject[])[0]!.status, "released"); + assert.equal((await executeCoordinationTodoUpdate(store, input)).status, "replayed"); + assert.deepEqual(await read(store), after); + const active = await seeded(provider, "2026-09-25T07:00:00Z"); + const before = await read(active); + assert.equal((await executeCoordinationTodoUpdate(active, input)).reason_code, "blocked_lifecycle_active_lease"); + assert.equal((await executeCoordinationTodoUpdate(active, {...input, + lease_idempotency_key: "old-execution", lease_expected_version: 29})).reason_code, + "blocked_lifecycle_execution_proof_not_allowed"); + assert.deepEqual(await read(active), before); + }); + providerTest(`${provider}: live lease, unauthorized actor and bundled work are rejected`, async () => { const active = await seeded(provider, "2026-09-25T07:00:00Z"); const before = await read(active); diff --git a/tests/control_plane_ts/explore_research_execution.test.ts b/tests/control_plane_ts/explore_research_execution.test.ts new file mode 100644 index 0000000000..39eb0fdb92 --- /dev/null +++ b/tests/control_plane_ts/explore_research_execution.test.ts @@ -0,0 +1,345 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import type {JsonObject} from "../../loopx/control_plane/effect_program.ts"; +import {normalizeResearchObservation, projectResearchFrontier, researchCompositionGaps} from "../../loopx/control_plane/capabilities/explore_research.ts"; +import {validateResearchExecution, normalizeResearchCompositionPolicy, projectResearchComposition, + researchCompositionFacts, qualifyResearchCompositionWriteback, + validateResearchCompositionSuccessor, qualifyResearchCompletion} from "../../loopx/control_plane/capabilities/explore_research_execution.ts"; + +function fixture(): JsonObject { + const observation = (node: string, target?: string): JsonObject => ({ + schema_version: "typed_research_observation_v0", explore_node_id: node, + progress: {schema_version: "typed_progress_observation_v0", work_item_id: `todo_${node}`, + result_class: "exploration_exhausted", coverage_scope_id: `scope-${node}`, + coverage_complete: true, evidence_ids: [`ev-${node}`]}, + closure_basis: {schema_version: "research_closure_basis_v0", disposition: "bounded", + constraints: [{kind: "invariant", id: "boundary", role: "decisive"}], evidence_ids: [`ev-${node}`]}, + composition_candidates: target ? [{target_node_id: target, basis: "explicit", + interaction_kind: "state_interference", evidence_ids: [`ev-${node}`, `ev-${target}`]}] : [], + }); + const nodes = ["a", "b"].map(node_id => ({node_id, node_kind: "hypothesis", status: "resolved", + research_observation: normalizeResearchObservation({observation: observation(node_id, node_id === "a" ? "b" : undefined)})})); + const params: JsonObject = {goal_id: "fixture", agent_id: "fixture-agent", nodes: [ + ...nodes, {node_id: "joint", node_kind: "experiment", status: "open", agent_id: "fixture-agent"}], + edges: ["a", "b"].map(to_node => ({from_node: "joint", to_node, edge_type: "depends_on"}))}; + const gap = researchCompositionGaps(params)[0]; + params.observation = {...observation("joint"), input_observations: gap.input_observations, + progress: {...observation("joint").progress as JsonObject, work_item_id: "todo_joint"}, + execution_lineage: {schema_version: "research_execution_lineage_v0", goal_id: "fixture", + gap_id: gap.gap_id, replan_obligation_id: "replan-0123456789abcdef", successor_todo_id: "todo_joint", agent_id: "fixture-agent"}}; + params.todo = {todo_id: "todo_joint", replan_obligation_id: "replan-0123456789abcdef", + claimed_by: "fixture-agent", task_class: "advancement_task", action_kind: "joint_probe", + explore_result_node_refs: ["joint"], target_key: "joint", actionable_open: true}; + return params; +} + +function liveFixture(): JsonObject { + const params = fixture(); + params.harness = {enabled: true, composition_mode: "explicit_only", composition_scope_id: "scope-joint"}; + params.todo = {...params.todo as JsonObject, status: "open"}; + const facts = researchCompositionFacts(params); + params.bindings = (facts.gaps as JsonObject[]).map(gap => ({gap_id: gap.gap_id, + obligation: {obligation_id: "replan-0123456789abcdef"}})); + params.todos = []; + const raw = params.observation as JsonObject; + params.observation = {...raw, progress: {...raw.progress as JsonObject, fingerprint: "progress-joint"}}; + return params; +} + +test("live policy is explicit, scoped and default off", () => { + assert.equal(normalizeResearchCompositionPolicy({harness: {enabled: true}}).enabled, false); + assert.equal(normalizeResearchCompositionPolicy({harness: {enabled: "true", composition_mode: "explicit_only", composition_scope_id: "scope"}}).enabled, false); + assert.throws(() => normalizeResearchCompositionPolicy({harness: {enabled: true, composition_mode: "explicit_only"}}), /scope/); + assert.throws(() => normalizeResearchCompositionPolicy({harness: {enabled: true, composition_mode: "infer"}}), /composition_mode/); + const params = liveFixture(); + params.harness = {enabled: true}; + assert.equal(projectResearchComposition(params).state, "disabled"); +}); + +test("only the exact runnable experiment successor schedules the live gap", () => { + const params = liveFixture(); + assert.equal(projectResearchComposition(params).pending_count, 1); + params.todos = [params.todo]; + const scheduled = projectResearchComposition(params); + assert.equal(scheduled.scheduled_count, 1); + assert.equal(scheduled.observed_count, 0); + for (const patch of [{claimed_by: "another-agent"}, {replan_obligation_id: "replan-fedcba9876543210"}, + {actionable_open: false, status: "deferred"}, {explore_result_node_refs: ["other"]}, {action_kind: "read"}]) { + assert.equal(projectResearchComposition({...params, todos: [{...params.todo as JsonObject, ...patch}]}).pending_count, 1); + } + const pending = projectResearchComposition({...params, todos: []}); + assert.equal(validateResearchCompositionSuccessor({frontier: pending, todo: params.todo, agent_id: params.agent_id}).accepted, true); + assert.throws(() => validateResearchCompositionSuccessor({frontier: pending, + todo: {...params.todo as JsonObject, status: "deferred", resume_when: "capacity_available:fixture"}, agent_id: params.agent_id}), /no deferral/); +}); + +test("live observation needs execution lineage and cannot be replaced by generic progress or a replay", () => { + const params = liveFixture(); + const pending = projectResearchComposition(params); + const gap = (pending.lineage_gaps as JsonObject[])[0]; + const guard = {schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore", + gap_id: gap.gap_id, frontier_revision: gap.frontier_revision}; + const qualify = (frontier: JsonObject, progress: JsonObject, repeated: string[] = []) => qualifyResearchCompositionWriteback({ + frontier, capability_guard: guard, obligation_id: "replan-0123456789abcdef", progress_observation: progress, + claimed_progress_fingerprints: repeated}); + assert.equal(qualify(pending, {fingerprint: "unrelated", result_class: "advanced"}).accepted, false); + const nodes = params.nodes as JsonObject[], raw = params.observation as JsonObject; + const unbound = {...raw}; delete unbound.execution_lineage; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", + research_observation: normalizeResearchObservation({observation: unbound})}]; + params.todos = [params.todo]; + assert.equal(projectResearchComposition(params).observed_count, 0); + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", + research_observation: normalizeResearchObservation({observation: raw})}]; + const observed = projectResearchComposition(params), progress = raw.progress as JsonObject; + assert.equal(observed.observed_count, 1); + assert.equal(qualify(observed, progress).capability_outcome, "composition_experiment_observed"); + assert.equal(qualify(observed, progress, ["progress-joint"]).accepted, false); + assert.equal(qualify(observed, {...progress, evidence_ids: []}).accepted, false); + assert.equal(qualify(observed, {...progress, work_item_id: "todo_other"}).accepted, false); + params.nodes = [{...nodes[0], status: "open"}, nodes[1], (params.nodes as JsonObject[])[2]]; + const invalidated = projectResearchComposition(params); + assert.equal(invalidated.ineligible_count, 1); + assert.equal(qualify(invalidated, progress).accepted, false); +}); + +test("Todo closeout needs this task's current terminal experiment result", () => { + const params = liveFixture(); + params.todos = [params.todo]; + const request = (frontier: JsonObject, todo = params.todo) => qualifyResearchCompletion({ + frontier, todo, actor_agent_id: params.agent_id, goal_id: params.goal_id}); + assert.equal(request(projectResearchComposition(params)).allowed, false); + const nodes = params.nodes as JsonObject[]; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", + research_observation: normalizeResearchObservation({observation: params.observation})}]; + const observed = projectResearchComposition(params); + const accepted = request(observed); + assert.equal(accepted.allowed, true); + assert.equal((accepted.evidence as JsonObject).todo_id, "todo_joint"); + assert.equal((accepted.evidence as JsonObject).experiment_node_id, "joint"); + assert.equal(request(observed, {...params.todo as JsonObject, todo_id: "todo_other"}).allowed, false); + params.nodes = [{...nodes[0], status: "open"}, nodes[1], (params.nodes as JsonObject[])[2]]; + assert.equal(request(projectResearchComposition(params)).allowed, false); + assert.equal(qualifyResearchCompletion({frontier: {enabled: false}, todo: params.todo}).required, false); +}); + +test("execution evidence binds Goal, gap, obligation, Todo, actor, experiment and current inputs", () => { + const params = fixture(); + const canonical = validateResearchExecution(params); + assert.deepEqual(canonical.execution_lineage, (params.observation as JsonObject).execution_lineage); + const unbound = {...params.observation as JsonObject}; + delete unbound.execution_lineage; + assert.notEqual(canonical.fingerprint, normalizeResearchObservation({observation: unbound}).fingerprint); + // The receipt and cold projection retain diagnostic semantics. No execution, + // scientific truth or Goal completion is inferred from validation alone. + assert.equal(projectResearchFrontier(params).mode, "read_only_shadow"); + assert.equal(researchCompositionGaps(params)[0].state, "pending"); +}); + +test("completed archive lineage retains evidence without making archived work runnable", () => { + const params = liveFixture(), nodes = params.nodes as JsonObject[]; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", + research_observation: normalizeResearchObservation({observation: params.observation})}]; + const retained = {...params.todo as JsonObject, status: "done", archive_state: "archive", actionable_open: false}; + for (const todo of [retained, {...retained, claimed_by: null}]) { + const frontier = projectResearchComposition({...params, todos: [todo]}); + assert.equal(frontier.observed_count, 1); + assert.equal(frontier.scheduled_count, 0); + assert.equal((frontier.lineage_gaps as JsonObject[])[0].observed_todo_id, "todo_joint"); + } + for (const patch of [{status: "open"}, {replan_obligation_id: "replan-fedcba9876543210"}, + {explore_result_node_refs: ["other"]}, {claimed_by: "other-agent"}]) { + assert.equal(projectResearchComposition({...params, todos: [{...retained, ...patch}]}).observed_count, 0); + } + assert.equal(projectResearchComposition({...params, todos: []}).observed_count, 0); + assert.equal(projectResearchComposition({...params, todos: [retained], + nodes: [{...nodes[0], status: "open"}, nodes[1], (params.nodes as JsonObject[])[2]]}).observed_count, 0); + assert.throws(() => validateResearchExecution({...params, todo: retained}), /runnable/); +}); + +test("evidence-backed candidate dismissal retires work without claiming an experiment outcome", () => { + const params = liveFixture(), nodes = params.nodes as JsonObject[]; + const raw = params.observation as JsonObject; + const dismissed = {...raw, progress: {...raw.progress as JsonObject, result_class: "no_followup"}, + closure_basis: {...raw.closure_basis as JsonObject, disposition: "no_followup"}, + composition_resolution: {schema_version: "research_composition_resolution_v0", disposition: "dismissed", + basis: "outside_scope", evidence_ids: ["ev-joint"]}}; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "dead_end", + research_observation: normalizeResearchObservation({observation: dismissed})}]; + params.todos = [params.todo]; + const frontier = projectResearchComposition(params), gap = (frontier.lineage_gaps as JsonObject[])[0]; + assert.equal(frontier.dismissed_count, 1); + assert.equal(frontier.observed_count, 0); + assert.equal(projectResearchFrontier(params).observed_count, 0); + const guard = {capability_id: "explore", gap_id: gap.gap_id, frontier_revision: gap.frontier_revision}; + const qualified = qualifyResearchCompositionWriteback({frontier, capability_guard: guard, + obligation_id: gap.obligation_id, progress_observation: dismissed.progress, claimed_progress_fingerprints: []}); + assert.equal(qualified.capability_outcome, "composition_candidate_dismissed"); + assert.equal(qualifyResearchCompletion({frontier, todo: params.todo, goal_id: params.goal_id, + actor_agent_id: params.agent_id}).allowed, true); + for (const patch of [{basis: "not_promising"}, {evidence_ids: []}, {evidence_ids: ["unrelated"]}]) { + assert.throws(() => normalizeResearchObservation({observation: {...dismissed, + composition_resolution: {...dismissed.composition_resolution, ...patch}}})); + } +}); + +test("temporary deferral needs a fresh exact blocker and the common resume contract", () => { + const params = liveFixture(), nodes = params.nodes as JsonObject[]; + const raw = params.observation as JsonObject; + const blocked = {...raw, progress: {...raw.progress as JsonObject, result_class: "blocked", + coverage_complete: false, blocker_id: "todo_dependency"}, closure_basis: null, + composition_resolution: {schema_version: "research_composition_resolution_v0", disposition: "deferred", + evidence_ids: ["ev-joint"]}}; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "blocked", + research_observation: normalizeResearchObservation({observation: blocked})}]; + const task = {...params.todo as JsonObject, status: "deferred", actionable_open: false, + resume_when: "todo_done:todo_dependency"}; + const blocker = {todo_id: "todo_dependency", role: "agent", task_class: "blocker", status: "open", + claimed_by: "fixture-agent", archive_state: "active", unblocks_todo_id: "todo_joint"}; + const deferred = projectResearchComposition({...params, todos: [task, blocker]}); + assert.equal(deferred.deferred_count, 1); + assert.equal(deferred.observed_count, 0); + assert.equal(deferred.pending_count, 0); + assert.equal(qualifyResearchCompletion({frontier: deferred, todo: task}).allowed, false); + const gap = (deferred.lineage_gaps as JsonObject[])[0]; + const request = {frontier: deferred, capability_guard: {capability_id: "explore", gap_id: gap.gap_id, + frontier_revision: gap.frontier_revision}, obligation_id: gap.obligation_id, + progress_observation: blocked.progress, claimed_progress_fingerprints: [], claimed_blocker_ids: []}; + assert.equal(qualifyResearchCompositionWriteback(request).capability_outcome, "composition_temporarily_deferred"); + assert.equal(qualifyResearchCompositionWriteback({...request, claimed_blocker_ids: ["todo_dependency"]}).accepted, false); + for (const patch of [{status: "done"}, {status: "blocked"}, {claimed_by: "other-agent"}, {task_class: "continuous_monitor"}, + {unblocks_todo_id: "todo_other"}, {archive_state: "archive"}]) { + assert.equal(projectResearchComposition({...params, todos: [task, {...blocker, ...patch}]}).deferred_count, 0); + } + for (const resume_when of ["capacity_available:fixture", "todo_done:todo_unknown", null]) { + assert.equal(projectResearchComposition({...params, todos: [{...task, resume_when}, blocker]}).deferred_count, 0); + } + // Resolving the prerequisite exposes the same open gap; it grants no lease + // and cannot turn the deferred declaration into an experimental outcome. + const resumed = projectResearchComposition({...params, todos: [task, {...blocker, status: "done"}]}); + assert.equal(resumed.pending_count, 1); + assert.equal(resumed.observed_count, 0); +}); + +test("a changed evidence duty needs explicit retirement facts and never closes live work", () => { + const params = liveFixture(), initial = projectResearchComposition(params); + const selected = (initial.lineage_gaps as JsonObject[])[0]; + const guard = {schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore", + gap_id: selected.gap_id, frontier_revision: selected.frontier_revision}; + const nodes = params.nodes as JsonObject[]; + const current = projectResearchComposition({...params, nodes: [{...nodes[0], status: "open"}, ...nodes.slice(1)]}); + const request = {frontier: current, capability_guard: guard, obligation_id: selected.obligation_id}; + const rejected = qualifyResearchCompositionWriteback(request); + assert.equal(rejected.accepted, false); + const contract = rejected.retirement_contract as JsonObject; + assert.equal(contract.disposition, "invalidated"); + const accepted = qualifyResearchCompositionWriteback({...request, + progress_observation: {...contract.progress_observation as JsonObject, fingerprint: "retirement-proof"}}); + assert.equal(accepted.capability_outcome, "composition_duty_invalidated"); + assert.deepEqual(accepted.outcomes, ["capability_duty_retired"]); + assert.equal(qualifyResearchCompositionWriteback({...request, frontier: initial, + progress_observation: contract.progress_observation}).accepted, false); + for (const patch of [{work_item_id: "replan-fedcba9876543210"}, {blocker_id: "unrelated"}, + {result_class: "advanced"}, {evidence_ids: ["unrelated"]}]) { + assert.equal(qualifyResearchCompositionWriteback({...request, + progress_observation: {...contract.progress_observation as JsonObject, ...patch}}).accepted, false); + } + const blocked = projectResearchComposition({...params, nodes: [{...nodes[0], status: "open"}, ...nodes.slice(1)], todos: [params.todo]}); + assert.equal(qualifyResearchCompositionWriteback({...request, frontier: blocked, + progress_observation: contract.progress_observation}).accepted, false); + assert.equal(qualifyResearchCompositionWriteback({...request, frontier: {schema_version: "unavailable"}, + progress_observation: contract.progress_observation}).retirement_contract, undefined); +}); + +test("rejected, deferred, wrong-agent and unrelated successors cannot attribute an execution result", () => { + for (const patch of [ + {actionable_open: false}, {claimed_by: "another-agent"}, {task_class: "continuous_monitor"}, + {action_kind: "inspect"}, {todo_id: "todo_other"}, {replan_obligation_id: "replan-fedcba9876543210"}, + {explore_result_node_refs: ["other"]}, {explore_result_node_refs: ["joint", "other"]}, + {target_key: "other"}, {archive_state: "archive"}, {excluded_agents: ["fixture-agent"]}, + ]) { + const params = fixture(); + params.todo = {...params.todo as JsonObject, ...patch}; + assert.throws(() => validateResearchExecution(params), /research execution/); + } + for (const patch of [{agent_id: "another-agent"}, {goal_id: "other-goal"}, {gap_id: "research-composition-0000000000000000"}]) { + const params = fixture(), raw = params.observation as JsonObject; + params.observation = {...raw, execution_lineage: {...raw.execution_lineage as JsonObject, ...patch}}; + assert.throws(() => validateResearchExecution(params), /research execution/); + } +}); + +test("stale fingerprints, reads, malformed lineage and already-observed inputs fail closed", () => { + const params = fixture(), raw = params.observation as JsonObject; + assert.throws(() => validateResearchExecution({...params, observation: {...raw, + input_observations: [{node_id: "a", fingerprint: "stale"}, ...(raw.input_observations as JsonObject[]).slice(1)]}}), /exact current/); + for (const progress of [{...raw.progress as JsonObject, result_class: "unchanged"}, + {...raw.progress as JsonObject, result_class: "advanced", evidence_ids: []}]) { + assert.throws(() => validateResearchExecution({...params, observation: {...raw, progress, closure_basis: null}}), /read or ACK/); + } + assert.throws(() => normalizeResearchObservation({observation: {...raw, + execution_lineage: {...raw.execution_lineage as JsonObject, successor_todo_id: "todo_other"}}}), /exact gap/); + const nodes = params.nodes as JsonObject[]; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", research_observation: normalizeResearchObservation({observation: raw})}]; + assert.equal(researchCompositionGaps(params)[0].state, "observed"); + assert.throws(() => validateResearchExecution(params), /pending gap/); +}); + +test("cold terminal diagnostics cannot discharge or block a live execution duty", () => { + const params = liveFixture(), nodes = params.nodes as JsonObject[], raw = params.observation as JsonObject; + const diagnostic: JsonObject = {...raw, explore_node_id: "diagnostic", + progress: {...raw.progress as JsonObject, work_item_id: "todo_diagnostic"}}; + delete diagnostic.execution_lineage; + params.nodes = [...nodes, {node_id: "diagnostic", node_kind: "experiment", status: "resolved", + research_observation: normalizeResearchObservation({observation: diagnostic})}]; + params.edges = [...params.edges as JsonObject[], ...["a", "b"].map(to_node => + ({from_node: "diagnostic", to_node, edge_type: "depends_on"}))]; + params.todos = [params.todo]; + assert.equal(researchCompositionGaps(params)[0].state, "observed"); + const scheduled = projectResearchComposition(params); + assert.equal(scheduled.scheduled_count, 1); + assert.equal(scheduled.observed_count, 0); + assert.doesNotThrow(() => validateResearchExecution({...params, frontier: scheduled})); + assert.throws(() => validateResearchExecution({...params, frontier: {...scheduled, agent_id: "another-agent"}}), /Goal and actor/); + assert.throws(() => validateResearchExecution({...params, frontier: {...scheduled, + policy: {...scheduled.policy as JsonObject, coverage_scope_id: "other-scope"}}}), /pending gap/); + params.nodes = [...(params.nodes as JsonObject[]).filter(node => node.node_id !== "joint"), + {...nodes[2], status: "resolved", research_observation: normalizeResearchObservation({observation: raw})}]; + const observed = projectResearchComposition(params); + assert.equal(observed.observed_count, 1); + assert.throws(() => validateResearchExecution({...params, frontier: observed}), /pending gap/); + const retained = {...params.todo as JsonObject, status: "done", claimed_by: null, archive_state: "archive", actionable_open: false}; + const retry = {...params.todo as JsonObject, todo_id: "todo_retry", explore_result_node_refs: ["retry"], target_key: "retry"}; + params.nodes = [...params.nodes as JsonObject[], {node_id: "retry", node_kind: "experiment", status: "resolved"}]; + params.edges = [...params.edges as JsonObject[], ...["a", "b"].map(to_node => + ({from_node: "retry", to_node, edge_type: "depends_on"}))]; + const retryRaw = {...raw, explore_node_id: "retry", progress: {...raw.progress as JsonObject, work_item_id: "todo_retry"}, + execution_lineage: {...raw.execution_lineage as JsonObject, successor_todo_id: "todo_retry"}}; + assert.throws(() => validateResearchExecution({...params, todo: retry, observation: retryRaw, + frontier: projectResearchComposition({...params, todos: [retained, retry]})}), /pending gap/); +}); + +test("the three-card presentation budget cannot hide execution attribution", () => { + const params = fixture(), nodes = params.nodes as JsonObject[]; + const template = nodes[0].research_observation as JsonObject; + for (const [source, target] of [["c", "d"], ["e", "f"], ["g", "h"]]) { + for (const node_id of [source, target]) { + const raw = {...template, explore_node_id: node_id, + progress: {...template.progress as JsonObject, work_item_id: `todo_${node_id}`, evidence_ids: [`ev-${node_id}`]}, + closure_basis: {...template.closure_basis as JsonObject, evidence_ids: [`ev-${node_id}`]}, + composition_candidates: node_id === source ? [{target_node_id: target, basis: "explicit", + interaction_kind: "state_interference", evidence_ids: [`ev-${source}`, `ev-${target}`]}] : []}; + nodes.push({node_id, node_kind: "hypothesis", status: "resolved", + research_observation: normalizeResearchObservation({observation: raw})}); + } + } + const all = researchCompositionGaps(params), compact = projectResearchFrontier(params); + assert.equal(all.length, 4); + assert.equal((compact.gaps as JsonObject[]).length, 3); + const omitted = all[3]; + params.edges = (omitted.input_node_ids as string[]).map(to_node => ({from_node: "joint", to_node, edge_type: "depends_on"})); + const raw = params.observation as JsonObject; + params.observation = {...raw, input_observations: omitted.input_observations, + execution_lineage: {...raw.execution_lineage as JsonObject, gap_id: omitted.gap_id}}; + assert.doesNotThrow(() => validateResearchExecution(params)); +}); diff --git a/tests/control_plane_ts/quota_settlement_readback.test.ts b/tests/control_plane_ts/quota_settlement_readback.test.ts index fd5304638b..3ddd6d856c 100644 --- a/tests/control_plane_ts/quota_settlement_readback.test.ts +++ b/tests/control_plane_ts/quota_settlement_readback.test.ts @@ -58,6 +58,78 @@ test("semantic replan guard distinguishes legacy, none, and exact selection", () ); }); +test("capability guard survives scalar receipt transport and rejects missing scope facts", () => { + const details = {semantic_replan_obligation_id: "replan-0000000000000001", + semantic_replan_capability_id: "explore", semantic_replan_gap_id: "research-composition-0123456789abcdef", + semantic_replan_frontier_revision: "research-composition-v0:0123456789abcdef"}; + assert.deepEqual(projectSemanticReplanGuard(details).selected_capability_guard, { + schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore", + gap_id: details.semantic_replan_gap_id, frontier_revision: details.semantic_replan_frontier_revision, + }); + assert.throws(() => projectSemanticReplanGuard({...details, semantic_replan_obligation_id: ""}), /selected obligation/); + assert.throws(() => projectSemanticReplanGuard({...details, semantic_replan_gap_id: ""}), /capability guard is malformed/); + assert.throws(() => projectSemanticReplanGuard({...details, semantic_replan_frontier_revision: undefined}), /capability guard is malformed/); +}); + +test("capability retirement verifies its exact receipt and never erases a committed debit", async () => { + const obligation = "replan-0000000000000001"; + const guard = {schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore", + gap_id: "research-composition-0123456789abcdef", frontier_revision: "research-composition-v0:0123456789abcdef"}; + const bound = settlementIdentity({goal_id: goalId, agent_id: agentId, todo_id: null, + replan_obligation_id: obligation, turn_instance_id: turnId}); + const progress = {schema_version: "typed_progress_observation_v0", work_item_id: obligation, + result_class: "blocked", blocker_id: "capability-invalidated-fixture", + evidence_ids: ["capability-evidence-v0:fedcba9876543210"], fingerprint: "retirement-progress"}; + const retirement = {schema_version: "capability_obligation_retirement_v0", disposition: "invalidated", + reason_code: "source_ineligible", capability_id: "explore", obligation_id: obligation, + original_guard: guard, current_revision: progress.evidence_ids[0], blocking_todo_count: 0, + blocking_todo_ids: [], progress_observation: progress, progress_fingerprint: progress.fingerprint}; + const ack = {recorded: true, source: "refresh_state_semantic_delta", semantic_delta: { + schema_version: "replan_semantic_delta_v0", accepted: true, obligation_id: obligation, + outcomes: ["capability_duty_retired"], satisfying_outcomes: ["capability_duty_retired"], capability_guard: guard, retirement}}; + const root = await fixture({writeback: true}); + try { + const index = join(root, "goals", goalId, "runs", "index.jsonl"); + const log = join(root, "goals", goalId, "rollout-event-log.jsonl"); + const events = (await readFile(log, "utf8")).trim().split("\n").map(line => JSON.parse(line)); + for (const event of events) { + event.details = {...event.details, todo_id: "", replan_obligation_id: obligation, settlement_effect_id: bound.effect_id}; + if (event.event_kind === "quota_should_run") Object.assign(event.details, { + semantic_replan_obligation_id: obligation, semantic_replan_capability_id: "explore", + semantic_replan_gap_id: guard.gap_id, semantic_replan_frontier_revision: guard.frontier_revision}); + } + await writeFile(log, events.map(event => JSON.stringify(event)).join("\n") + "\n"); + const base = {classification: "state_refreshed", goal_id: goalId, agent_id: agentId, + turn_instance_id: turnId, replan_obligation_id: obligation, settlement_identity: bound, + delivery_outcome: "outcome_gap", progress_observation: progress, autonomous_replan_ack: ack}; + const read = () => readQuotaSettlement(request(root, {todo_id: null, replan_obligation_id: obligation})); + await writeFile(index, JSON.stringify(base) + "\n"); + const accepted = await read(); + assert.equal((accepted.progress as any).closeout_kind, "capability_duty_retired_no_spend"); + assert.equal(accepted.replay_phase, "settled"); + assert.equal((accepted.spend as any).payload.ok, false); + assert.equal((accepted.terminal_closeout as any).payload.ok, false); + for (const patch of [{obligation_id: "replan-0000000000000002"}, {blocking_todo_count: 1}, + {original_guard: {...guard, frontier_revision: "research-composition-v0:fedcba9876543210"}}, + {current_revision: "unrelated"}, {progress_fingerprint: "forged"}]) { + const changed = {...base, autonomous_replan_ack: {...ack, semantic_delta: {...ack.semantic_delta, + retirement: {...retirement, ...patch}}}}; + await writeFile(index, JSON.stringify(changed) + "\n"); + assert.equal((await read()).replay_phase, "settlement_pending"); + } + await writeFile(index, JSON.stringify(base) + "\n" + JSON.stringify({classification: "quota_slot_spent", + goal_id: goalId, agent_id: agentId, turn_instance_id: turnId, replan_obligation_id: obligation, + settlement_identity: bound}) + "\n"); + await appendFile(log, JSON.stringify({schema_version: "loopx_rollout_event_v0", event_id: "spent-before-retirement", + event_kind: "quota_spend", goal_id: goalId, agent_id: agentId, run_id: turnId, + details: {settlement_effect_id: bound.effect_id}}) + "\n"); + const spent = await read(); + assert.equal((spent.progress as any).state, "settled"); + assert.equal((spent.progress as any).closeout_kind, undefined); + assert.equal((spent.spend as any).payload.ok, true); + } finally {await rm(root, {recursive: true, force: true});} +}); + async function fixture(options: { guard?: boolean; /** diff --git a/tests/control_plane_ts/replan_semantics.test.ts b/tests/control_plane_ts/replan_semantics.test.ts index f53f85890f..590f7076f2 100644 --- a/tests/control_plane_ts/replan_semantics.test.ts +++ b/tests/control_plane_ts/replan_semantics.test.ts @@ -1,6 +1,18 @@ import assert from "node:assert/strict"; import test from "node:test"; import { projectReplanSemantics, requiredSemanticOutcomes } from "../../loopx/control_plane/work_items/replan_semantics.ts"; + +test("capability evidence is not a default exit or a generic progress claim", () => { + for (const outcome of ["capability_evidence_observed", "capability_duty_retired"]) { + assert.equal(requiredSemanticOutcomes({triggers: [{kind: "no_progress_streak"}]}).includes("capability_evidence_observed"), false); + assert.throws(() => requiredSemanticOutcomes({satisfying_semantic_outcomes: ["capability_evidence_observed"]}), /bound owning capability/); + const obligation = {satisfying_semantic_outcomes: [outcome], + capability_guard: {schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore"}}; + assert.deepEqual(requiredSemanticOutcomes(obligation), [outcome]); + assert.throws(() => projectReplanSemantics({operation: "qualify", obligation, + observation_delta: {delta_kinds: [outcome]}}), /owning capability, not generic progress/); + } +}); import { visionAuthoringContract } from "../../loopx/control_plane/goals/vision_checkpoint.ts"; import type { JsonObject } from "../../loopx/control_plane/effect_program.ts"; diff --git a/tests/test_chat_goal_configuration_api.py b/tests/test_chat_goal_configuration_api.py index a14964e1a4..158c7d49e9 100644 --- a/tests/test_chat_goal_configuration_api.py +++ b/tests/test_chat_goal_configuration_api.py @@ -1,6 +1,7 @@ from __future__ import annotations from pathlib import Path +import json from types import SimpleNamespace from typing import Any @@ -54,6 +55,18 @@ def _catalog_payload(*, explore_enabled: bool = False) -> dict[str, Any]: } +def test_explore_composition_options_keep_typed_scope_and_reject_coercion() -> None: + options = _goal_capability_options("explore_harness", { + "enabled": True, "composition_mode": "explicit_only", "composition_scope_id": "joint-scope", + }) + assert options["explore_composition_mode"] == "explicit_only" + assert options["explore_composition_scope_id"] == "joint-scope" + assert _goal_capability_options("explore_harness", {"enabled": True, "composition_mode": ""})["explore_composition_mode"] == "disabled" + for field, value in [("composition_mode", True), ("composition_scope_id", 0), ("composition_scope_id", [])]: + with pytest.raises(TypeError, match="must be strings"): + _goal_capability_options("explore_harness", {"enabled": True, field: value}) + + def _periodic_catalog_payload(*, override_present: bool) -> dict[str, Any]: feature: dict[str, Any] = { "feature_id": "periodic_report", @@ -421,6 +434,57 @@ def test_goal_configuration_apply_rejects_stale_preview() -> None: assert handler.responses[0]["error_code"] == "goal_configuration_preview_stale" +def test_research_policy_api_applies_and_reads_real_source_and_shared_registry(tmp_path: Path) -> None: + project, runtime, shared = tmp_path / "project", tmp_path / "runtime", tmp_path / "shared" + registry = project / ".loopx" / "registry.json" + registry.parent.mkdir(parents=True) + registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{ + "id": "goal-example", "repo": str(project), "status": "active", + "spawn_policy": {"explore_harness": {"enabled": False}}, + }]})) + + class RealHandler(_MutationHandler): + _goal_configuration_reader = GoalConfigurationRequestMixin._goal_configuration_reader + _goal_configuration_writer = GoalConfigurationRequestMixin._goal_configuration_writer + + def __init__(self, body): + super().__init__(body) + self.server = SimpleNamespace(registry_path=registry, runtime_root=runtime, + runtime_root_override=str(shared)) + + def _goal_configuration_machine_namespaces(self): + return [] + + for configuration in [ + {"enabled": True, "composition_mode": "explicit_only", "composition_scope_id": "joint-scope"}, + {"enabled": True, "composition_mode": "disabled", "composition_scope_id": "joint-scope"}, + ]: + body = {"goal_id": "goal-example", "capability_id": "explore_harness", "configuration": configuration} + before = registry.read_bytes() + preview = RealHandler(body) + preview._goal_configuration_update(execute=False) + plan = preview.responses[0] + assert plan["status_code"] == 201, plan + assert registry.read_bytes() == before + applied = RealHandler({**body, "expected_plan_revision": plan["plan_revision"]}) + applied.path = CHAT_GOAL_CONFIGURATION_APPLY_PATH + applied._goal_configuration_update(execute=True) + receipt = applied.responses[0] + assert receipt["status_code"] == 200, receipt + assert receipt["readback_verified"] is True + assert {key: receipt["goal_configuration"][key] for key in configuration} == configuration + source = json.loads(registry.read_text())["goals"][0]["spawn_policy"]["explore_harness"] + mirror = json.loads((shared / "registry.global.json").read_text())["goals"][0]["spawn_policy"]["explore_harness"] + assert {key: source[key] for key in configuration} == configuration + assert mirror == source + stale = RealHandler({**body, "expected_plan_revision": plan["plan_revision"]}) + stale.path = CHAT_GOAL_CONFIGURATION_APPLY_PATH + after = registry.read_bytes(), (shared / "registry.global.json").read_bytes() + stale._goal_configuration_update(execute=True) + assert stale.responses[0]["status_code"] == 409 + assert (registry.read_bytes(), (shared / "registry.global.json").read_bytes()) == after + + def test_goal_configuration_clear_override_is_revision_locked() -> None: body = { "goal_id": "goal-example", diff --git a/tsconfig.control-plane.json b/tsconfig.control-plane.json index c928bba0f7..8b0618fa64 100644 --- a/tsconfig.control-plane.json +++ b/tsconfig.control-plane.json @@ -15,6 +15,8 @@ "include": [ "loopx/control_plane/capabilities/explore_research.ts", "tests/control_plane_ts/explore_research.test.ts", + "loopx/control_plane/capabilities/explore_research_execution.ts", + "tests/control_plane_ts/explore_research_execution.test.ts", "loopx/control_plane/runtime/usage_statistics*.ts", "tests/control_plane_ts/usage_statistics.test.ts", "tests/control_plane_ts/usage_statistics_delivery.test.ts",