From f8f49525ee47a752acb93ddc8ce1161f72983f5f Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Mon, 28 Sep 2026 21:35:46 -0700 Subject: [PATCH 01/12] feat(explore): enforce scoped composition lineage and lifecycle evidence Signed-off-by: Lihua <1017343802@qq.com> --- .../capability-configuration-fields.tsx | 4 +- .../capability-localization.ts | 10 +- .../goal-capability-settings.tsx | 6 +- loopx/capabilities/configuration_ui.py | 4 + .../explore/composition_frontier.py | 17 + .../capabilities/explore/research_evidence.py | 34 ++ .../capabilities/explore/research_frontier.py | 218 ++++++++++++ .../explore/research_snapshot_host.py | 45 +++ loopx/chat_goal_configuration_api.py | 8 +- loopx/cli_commands/explore.py | 40 ++- loopx/cli_commands/explore_feishu_commands.py | 2 + .../cli_commands/project_lifecycle_inputs.py | 1 + .../project_lifecycle_refresh_state.py | 1 + loopx/cli_commands/registry_admin.py | 2 + .../cli_commands/registry_admin_configure.py | 8 + loopx/cli_commands/status.py | 25 +- loopx/cli_commands/todo.py | 15 + loopx/cli_commands/turn.py | 1 + loopx/configuration_catalog.py | 4 +- loopx/configure_goal.py | 15 + .../capabilities/explore_research.ts | 73 +++- .../explore_research_execution.ts | 321 ++++++++++++++++++ .../capabilities/explore_research_terminal.ts | 93 +++++ .../coordination/local_authority_runtime.ts | 6 +- .../coordination/todo_blocked_lifecycle.ts | 14 +- .../coordination/todo_terminal_lifecycle.ts | 20 ++ loopx/control_plane/effect_runtime.py | 7 + .../control_plane/effect_runtime_handlers.ts | 10 + .../goals/goal_frontier/__init__.py | 13 +- .../goals/goal_frontier/replan_rules.py | 10 + .../control_plane/quota/heartbeat_receipt.py | 20 ++ loopx/control_plane/quota/settlement.py | 16 +- loopx/control_plane/quota/settlement_cli.py | 9 + loopx/control_plane/quota/settlement_phase.ts | 39 +++ .../quota/settlement_readback.ts | 45 ++- .../control_plane/quota/should_run_packet.py | 8 +- loopx/control_plane/quota/slot_accounting.py | 4 +- .../quota/unsettled_host_turn_recovery.ts | 5 +- .../status/agent_lane_projection.py | 1 + .../work_items/progress_observation.py | 5 + .../work_items/replan_semantics.ts | 13 +- .../work_items/semantic_replan_writeback.py | 38 ++- loopx/orchestration.py | 6 + .../presentation/renderers/status_markdown.py | 9 + loopx/quota.py | 8 +- .../project_registry_io_manifest_v1.json | 32 +- loopx/state_refresh.py | 16 +- loopx/todos.py | 13 + 48 files changed, 1243 insertions(+), 71 deletions(-) create mode 100644 loopx/capabilities/explore/research_frontier.py create mode 100644 loopx/capabilities/explore/research_snapshot_host.py create mode 100644 loopx/control_plane/capabilities/explore_research_execution.ts create mode 100644 loopx/control_plane/capabilities/explore_research_terminal.ts diff --git a/apps/presentation/dashboard/src/features/personal-workspace/capability-configuration-fields.tsx b/apps/presentation/dashboard/src/features/personal-workspace/capability-configuration-fields.tsx index eeea715829..7134ff18b3 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/capability-configuration-fields.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/capability-configuration-fields.tsx @@ -3,7 +3,7 @@ import { useId, type ReactNode } from "react"; import type { CapabilityConfigurationEditor } from "../../data/chat"; import { PeriodicReportScheduleField } from "./periodic-report-schedule-field"; -type FieldCopy = Record; +type FieldCopy = Record }>; type ConfigurationField = CapabilityConfigurationEditor["fields"][number]; type FieldValue = boolean | number | string | string[] | Record | null; type FieldChange = (key: string, value: FieldValue) => void; @@ -40,7 +40,7 @@ function ConfigurationFieldControl({ copy, field, id, onChange, value, timezone {label} ); diff --git a/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts b/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts index 0756239373..25c50ed772 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts +++ b/apps/presentation/dashboard/src/features/personal-workspace/capability-localization.ts @@ -8,7 +8,7 @@ type LocalizedCopy = Readonly<{ readOnlyReason?: string; }>; -type FieldCopy = Record>; +type FieldCopy = Record }>>; const capabilityCopy: Record> = { en: { @@ -36,7 +36,7 @@ const capabilityCopy: Record> = { }, explore_harness: { displayName: "Explore Harness", - description: "Selects a capability-owned planning and research harness profile for bounded multi-step exploration.", + description: "Plans bounded exploration. Explicit composition replans require a bound experiment or typed result; task completion and execution permissions remain separate.", }, lark_event_inbox: { displayName: "Lark event inbox", @@ -102,7 +102,7 @@ const capabilityCopy: Record> = { }, explore_harness: { displayName: "探索 Harness", - description: "为有界的多步探索选择由能力负责的规划与研究 Harness profile。", + description: "规划有界探索。显式组合的重新规划需绑定实验或类型化结果;任务完成和执行权限单独判断。", }, lark_event_inbox: { displayName: "飞书事件收件箱", @@ -162,6 +162,8 @@ const fieldCopy: Record = { reasoning_effort: { label: "Child reasoning effort", description: "For example max; the host must support this model and effort." }, max_children: { label: "Maximum children", description: "Hard upper bound for concurrently delegated child work." }, profile: { label: "Planner profile", description: "Select one registered Explore Harness profile." }, + composition_mode: { label: "Composition policy", description: "Replan requires an exact experiment successor or typed result; it grants no execution authority.", optionLabels: {disabled: "Off", explicit_only: "Explicit candidates"} }, + composition_scope_id: { label: "Research coverage scope", description: "An opaque scope id required by the explicit-only composition policy." }, profile_preset: { label: "Report profile", description: "Capability-owned report profile, such as weekly-progress." }, wait_for_ci: { label: "Wait for CI", description: "Disable to use local validation without querying or waiting for CI. Merge authority is unchanged." }, review_priority: { label: "Review priority", description: "Choose whether other developers' PRs or the authenticated reviewer's own PRs are ranked first." }, @@ -189,6 +191,8 @@ const fieldCopy: Record = { reasoning_effort: { label: "子 Agent 推理档位", description: "例如 max;宿主须支持所选模型与档位。" }, max_children: { label: "最大子 Agent 数", description: "可同时委派的子任务硬上限。" }, profile: { label: "规划 Profile", description: "选择一个已注册的 Explore Harness profile。" }, + composition_mode: { label: "组合策略", description: "重新规划需要精确的实验后继或类型化结果;它不授予执行权限。", optionLabels: {disabled: "关闭", explicit_only: "仅显式候选"} }, + composition_scope_id: { label: "研究覆盖范围", description: "显式组合策略要求填写不含私有内容的范围标识。" }, profile_preset: { label: "报告 Profile", description: "由该能力管理的报告 profile,例如 weekly-progress。" }, wait_for_ci: { label: "等待 CI", description: "关闭后使用本地验证,不查询或等待 CI;不改变合并权限。" }, review_priority: { label: "审阅优先级", description: "选择先排其他开发者的 PR,还是先排当前已认证审阅者自己的 PR。" }, diff --git a/apps/presentation/dashboard/src/features/personal-workspace/goal-capability-settings.tsx b/apps/presentation/dashboard/src/features/personal-workspace/goal-capability-settings.tsx index d4e48be011..cb7c40dabc 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/goal-capability-settings.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/goal-capability-settings.tsx @@ -1,4 +1,4 @@ -import { useEffect, useMemo, useState } from "react"; +import { useEffect, useLayoutEffect, useMemo, useState } from "react"; import { AlertTriangle, Code2, LoaderCircle, RefreshCw } from "lucide-react"; import { @@ -54,7 +54,9 @@ function useCapabilityMutation({ goalId, onApplied, selected, t }: Readonly<{ ? parseEditableCapabilityJson(selected.configuration_editor, jsonDraft) : null, [selected, jsonDraft]); const jsonValid = editorMode === "guided" || parsedJson !== null; - useEffect(() => { + // Initialize the new capability's draft before it can receive input. A + // post-paint reset can otherwise erase the first toggle after selection. + useLayoutEffect(() => { setEditorMode("guided"); setJsonDraft(""); setMutation({ diff --git a/loopx/capabilities/configuration_ui.py b/loopx/capabilities/configuration_ui.py index c568294335..642c55e0b2 100644 --- a/loopx/capabilities/configuration_ui.py +++ b/loopx/capabilities/configuration_ui.py @@ -290,6 +290,10 @@ def capability_configuration_editor( "select", options=explore_harness_profiles, ), + _field("composition_mode", "Composition policy", "select", options=["disabled", "explicit_only"], + description="Replan requires an exact experiment successor or typed result; it grants no execution authority."), + _field("composition_scope_id", "Research coverage scope", "text", nullable=True, + description="An opaque scope id required by the explicit-only composition policy."), ], }, "change_quality_qualification": { diff --git a/loopx/capabilities/explore/composition_frontier.py b/loopx/capabilities/explore/composition_frontier.py index cdfe6c3917..0a47b72742 100644 --- a/loopx/capabilities/explore/composition_frontier.py +++ b/loopx/capabilities/explore/composition_frontier.py @@ -213,6 +213,8 @@ def project_live_explore_composition_frontier( goal_id: str, agent_id: str | None, status_payload: Mapping[str, Any], + state_text: str | None = None, + capability_guard: Mapping[str, Any] | None = None, ) -> dict[str, Any] | None: """Read the goal's Explore graph for one live quota decision. @@ -241,6 +243,12 @@ def project_live_explore_composition_frontier( ) enabled = harness.get("enabled") is True log_path = explore_result_log_path(runtime_root, goal_id) + pinned_research = (capability_guard or {}).get("capability_id") == "explore" + if pinned_research and (not enabled or harness.get("composition_mode") != "explicit_only"): + from .research_frontier import build_research_composition_frontier, read_research_todo_history + return build_research_composition_frontier({"goal_id": goal_id, "nodes": [], "edges": []}, + candidate_sources=[], harness=harness, todos=read_research_todo_history( + runtime_root=runtime_root, goal=dict(goal), state_text=state_text), agent_id=agent_id) if not enabled: return None try: @@ -283,6 +291,15 @@ def project_live_explore_composition_frontier( if isinstance(project_asset.get("agent_todos"), Mapping) else {} ) + if harness.get("composition_mode") == "explicit_only": + from .research_frontier import build_research_composition_frontier, read_research_todo_history + return build_research_composition_frontier( + projection, candidate_sources=[ + {"node_id": event["result_id"], "research_observation": event["research_observation"]} + for event in events if event.get("research_observation") + ], harness=harness, todos=read_research_todo_history( + runtime_root=runtime_root, goal=dict(goal), state_text=state_text), agent_id=agent_id, + ) return build_explore_composition_frontier( projection, todos=_todo_items(item_todos, asset_todos), diff --git a/loopx/capabilities/explore/research_evidence.py b/loopx/capabilities/explore/research_evidence.py index a52ef6431e..6cff3c9703 100644 --- a/loopx/capabilities/explore/research_evidence.py +++ b/loopx/capabilities/explore/research_evidence.py @@ -22,6 +22,16 @@ def _research_result(method: str, params: Mapping[str, Any]) -> dict[str, Any]: return result +def _research_todo_facts(todo: Mapping[str, Any]) -> dict[str, Any]: + """Bound local authority facts; never transport private task text/notes.""" + from ...control_plane.todos.todo_semantics import todo_item_is_actionable_open + fields = ("todo_id", "role", "status", "claimed_by", "replan_obligation_id", "task_class", + "action_kind", "archive_state", "excluded_agents", "explore_result_node_refs", "target_key", "capability_binding_ref", + "resume_when", "unblocks_todo_id") + return {**{field: todo[field] for field in fields if field in todo}, + "actionable_open": todo_item_is_actionable_open(dict(todo))} + + def normalize_research_observation(value: Mapping[str, Any]) -> dict[str, Any]: validate_public_safe_value(value, path="research_observation") raw = dict(value) @@ -58,6 +68,8 @@ def validate_research_append( if dict(event) != previous: raise ValueError("research observation replay cannot rewrite its node revision; use explore observe") return + if observation.get("execution_lineage"): + raise ValueError("new research execution lineage requires explore observe and a current canonical Todo read") projection = proposal or build_explore_result_projection([*events, event], goal_id=str(event["goal_id"])) _research_result("explore.research.validate_attribution", { "observation": observation, "nodes": projection["nodes"], "edges": projection["edges"], @@ -66,6 +78,7 @@ def validate_research_append( def append_research_observation( path: Path, *, goal_id: str, observation: Mapping[str, Any], agent_id: str | None = None, + registry_path: Path | None = None, runtime_root: Path | None = None, ) -> dict[str, Any]: # Use the existing append-only node revision transport. Lock attribution, # replay and append together so an input update cannot slip between them. @@ -88,6 +101,27 @@ def append_research_observation( _research_result("explore.research.validate_attribution", { "observation": canonical, "nodes": projection["nodes"], "edges": projection["edges"], }) + if canonical.get("execution_lineage"): + if registry_path is None or runtime_root is None: + raise ValueError("research execution lineage requires the selected registry and source runtime") + from ...todos import list_goal_todos + + lineage = canonical["execution_lineage"] + context = list_goal_todos( + registry_path=registry_path, runtime_root_arg=str(runtime_root), goal_id=goal_id, + role="agent", todo_id=lineage["successor_todo_id"], + ) + todos = context.get("todos") or [] + todo = todos[0] if len(todos) == 1 else {} + _research_result("explore.research.validate_execution", { + "goal_id": goal_id, "agent_id": agent_id, "observation": canonical, + "nodes": projection["nodes"], "edges": projection["edges"], + "candidate_sources": [ + {"node_id": event["result_id"], "research_observation": event["research_observation"]} + for event in events if event.get("research_observation") + ], + "todo": _research_todo_facts(todo), + }) event = build_explore_node_event( goal_id=goal_id, title=node["title"], node_id=node["node_id"], node_kind=node["node_kind"], status=node["status"], summary=node["summary"], blocked_reason=node["blocked_reason"], diff --git a/loopx/capabilities/explore/research_frontier.py b/loopx/capabilities/explore/research_frontier.py new file mode 100644 index 0000000000..b052946437 --- /dev/null +++ b/loopx/capabilities/explore/research_frontier.py @@ -0,0 +1,218 @@ +"""Live Explore transport. TypeScript owns policy, eligibility and lineage. + +The common replan owner retains obligation identity. No provider mutation or +work authority is performed while projecting research evidence. +""" +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from contextlib import contextmanager +from pathlib import Path +from typing import Any + +from ...control_plane.work_items.autonomous_replan_obligation import build_autonomous_replan_obligation_payload +from .research_evidence import _research_result, _research_todo_facts + + +def read_research_todo_history( + *, runtime_root: Path, goal: dict[str, Any], state_text: str | None = None, +) -> list[dict[str, Any]]: + """Read canonical active and retained rows, never a compact display list. + + The existing Todo codecs supply authority and lifecycle facts. Archives + are evidence lineage only; the typed owner decides their eligibility. + """ + from ...control_plane.coordination.local_authority import canonical_todo_items, read_canonical_todos_if_promoted + from ...control_plane.todos.active_state_todo_parser import parse_active_state_todos, parse_todo_source + from ...materials import goal_state_path + + canonical = read_canonical_todos_if_promoted(runtime_root=runtime_root, goal_id=goal["id"]) + if canonical is not None: + guards = canonical.get("goal_acceptance_work_guards") or {} + return [{**todo, **({"goal_acceptance_guard": guards[todo["todo_id"]]} if todo["todo_id"] in guards else {})} + for todo in canonical_todo_items(canonical["todos"]) if todo.get("role") != "user"] + if state_text is None: + path = goal_state_path(goal) + if path is None: + raise ValueError("research lineage requires the Goal's canonical Todo source") + state_text = path.read_text(encoding="utf-8") + active = parse_active_state_todos(state_text, item_limit=None, goal=goal).get("agent_todos", {}).get("items", []) + _, archived, _ = parse_todo_source(state_text, goal=goal) + return [*active, *[todo for todo in archived if todo.get("role") != "user"]] + + +def research_composition_obligation(gap: Mapping[str, Any], *, agent_id: str | None) -> dict[str, Any]: + guard = {"schema_version": "semantic_replan_capability_guard_v0", "capability_id": "explore", + "gap_id": gap["gap_id"], "frontier_revision": gap["frontier_revision"]} + return build_autonomous_replan_obligation_payload( + schema_version="autonomous_replan_obligation_v0", agent_id=agent_id, include_agent_id=True, + stall_threshold=0, trigger_count=1, + triggers=[{"kind": "capability_evidence_gap", "capability_id": "explore", "agent_id": agent_id, + "frontier_identity": gap["gap_id"], "frontier_revision": gap["frontier_revision"], + "text": "An explicit evidence gap requires a bound experiment and typed result."}], + guidance_actions=["create_successor", "record_typed_result"], + todo_actions=[{"action": "add", "role": "agent", "priority": "P1", + "text": gap.get("successor_summary") or "Run a bounded experiment for the explicit input set."}], + stop_condition="Respect the Goal's existing authority, budget and protected-operation gates.", + recommended_action="Bind a runnable experiment to the exact evidence gap or record its typed, attributable result.", + extra_fields={"capability_guard": guard, "satisfying_semantic_outcomes": [ + "new_runnable_successor", "capability_evidence_observed", "new_concrete_blocker", "capability_duty_retired"]}, + ) + + +def build_research_composition_frontier( + projection: Mapping[str, Any], *, candidate_sources: list[dict[str, Any]], + harness: Mapping[str, Any], todos: Sequence[Mapping[str, Any]], agent_id: str | None, +) -> dict[str, Any]: + params = {"goal_id": projection["goal_id"], "nodes": projection["nodes"], "edges": projection["edges"], + "candidate_sources": candidate_sources, "harness": dict(harness), "agent_id": agent_id} + facts = _research_result("explore.research.composition_facts", params) + bindings = [{"gap_id": gap["gap_id"], "obligation": research_composition_obligation(gap, agent_id=agent_id)} + for gap in facts["gaps"]] + frontier = _research_result("explore.research.composition", {**params, "bindings": bindings, + "todos": [_research_todo_facts(todo) for todo in todos]}) + by_gap = {binding["gap_id"]: binding["obligation"] for binding in bindings} + # These internal candidates are needed for the original Turn's settlement + # after its exact successor changes the actionable frontier. They are not + # added to the public quota packet. + transitions = [] + for gap in frontier["lineage_gaps"]: + obligation = by_gap[gap["gap_id"]] + if gap["status"] == "scheduled": + delta = {"schema_version": "replan_semantic_delta_v0", "accepted": True, + "obligation_id": obligation["obligation_id"], "outcomes": ["new_runnable_successor"], + "satisfying_outcomes": ["new_runnable_successor"], "successor_todo_id": gap["successor_todo_id"], + "capability_guard": obligation["capability_guard"], + "capability_outcome": "new_runnable_composition_experiment"} + transitions.append({"obligation": obligation, "ack": { + "schema_version": "autonomous_replan_ack_v0", "recorded": True, + "source": "todo_replan_successor_transition", "semantic_delta": delta}}) + frontier["settlement_transitions"] = transitions + selected = frontier.get("selected_gap") + frontier["obligation"] = by_gap[selected["gap_id"]] if selected else None + # Compact guard facts include every candidate, so invalidation and omitted + # display cards cannot be mistaken for a missing settled obligation. + frontier["guard_facts"] = [{"capability_guard": binding["obligation"]["capability_guard"], + "obligation_id": binding["obligation"]["obligation_id"]} for binding in bindings] + frontier["public_projection"] = public_research_composition_frontier(frontier) + return frontier + + +def public_research_composition_frontier(frontier: Mapping[str, Any] | None) -> dict[str, Any] | None: + if frontier is None: + return None + public = {key: value for key, value in frontier.items() + if key not in {"lineage_gaps", "settlement_transitions", "guard_facts", "obligation", "obligation_tasks", "public_projection"}} + if frontier.get("schema_version") == "research_composition_frontier_v0": + public["gaps"] = [{key: value for key, value in gap.items() if key != "execution_results"} + for gap in frontier["gaps"]] + if frontier.get("selected_gap"): + public["selected_gap"] = {key: value for key, value in frontier["selected_gap"].items() if key != "execution_results"} + return public + + +def prepare_research_replan_evidence( + *, runtime_root: Path, goal_id: str, agent_id: str, registry_goal: dict[str, Any] | None, + state_text: str, capability_guard: Mapping[str, Any] | None = None, +): + """IO composition root adapter; the shared gate never imports Explore.""" + import shlex + from ...control_plane.work_items.semantic_replan_writeback import CapabilityReplanEvidence + from .composition_frontier import project_live_explore_composition_frontier + + goal = {**(registry_goal or {}), "id": goal_id} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + if not (harness.get("enabled") is True and harness.get("composition_mode") == "explicit_only") and capability_guard is None: + return None + frontier = project_live_explore_composition_frontier(runtime_root=runtime_root, goal_id=goal_id, + agent_id=agent_id, state_text=state_text, capability_guard=capability_guard, + status_payload={"run_history": {"goals": [goal]}}) or {} + + def qualify(guard: Mapping[str, Any], obligation_id: str, progress: dict[str, Any] | None, runs: list[dict[str, Any]]): + if guard.get("capability_id") != "explore": + raise ValueError("selected capability writeback owner is not available") + obligation = research_composition_obligation( + {"gap_id": guard["gap_id"], "frontier_revision": guard["frontier_revision"]}, agent_id=agent_id) + if obligation["obligation_id"] != obligation_id: + raise ValueError("selected capability guard does not match its obligation identity") + delta = _research_result("explore.research.composition_writeback", {"frontier": frontier, + "capability_guard": dict(guard), "obligation_id": obligation_id, "progress_observation": progress, + "claimed_progress_fingerprints": [(run.get("progress_observation") or {}).get("fingerprint") for run in runs], + "claimed_blocker_ids": [(run.get("progress_observation") or {}).get("blocker_id") for run in runs]}) + delta["readback_actions"] = [f"loopx --runtime-root {shlex.quote(str(runtime_root))} explore summary " + f"--goal-id {shlex.quote(goal_id)} --agent-id {shlex.quote(agent_id)} --format json"] + return obligation, delta + return CapabilityReplanEvidence(frontier=frontier, qualify=qualify) + + +def attach_research_execution_projection( + projection: dict[str, Any], *, events: list[dict[str, Any]], registry: dict[str, Any], + runtime_root: Path, agent_id: str | None, +) -> None: + """Render the same live facts through existing Explore and sink fields.""" + from ...agent_registry import registered_agent_ids_for_goal + from ...materials import find_registry_goal + from ...control_plane.todos.contract import normalize_todo_claimed_by + + goal = find_registry_goal(registry, projection["goal_id"]) or {} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + if harness.get("enabled") is not True or harness.get("composition_mode") != "explicit_only": + return + registered = registered_agent_ids_for_goal(goal) + actor = normalize_todo_claimed_by(agent_id) if agent_id else registered[0] if len(registered) == 1 else None + if actor is None or actor not in registered: + raise ValueError("Live research presentation requires --agent-id naming one registered Goal agent") + frontier = build_research_composition_frontier(projection, + candidate_sources=[{"node_id": e["result_id"], "research_observation": e["research_observation"]} + for e in events if e.get("research_observation")], + harness=harness, todos=read_research_todo_history(runtime_root=runtime_root, goal=goal), agent_id=actor) + projection["research_execution_frontier"] = public_research_composition_frontier(frontier) + by_node: dict[str, list[dict[str, Any]]] = {} + for gap in frontier["lineage_gaps"]: + for node_id in [*gap["input_node_ids"], *gap["experiment_node_ids"]]: + by_node.setdefault(node_id, []).append(gap) + for node in projection["nodes"]: + gaps = by_node.get(node["node_id"], []) + if not gaps: + continue + facts = [f"Execution composition ({actor}): {gap['gap_id']} {gap['status']}" + for gap in gaps[:3]] + if len(gaps) > 3: + facts.append(f"{len(gaps) - 3} additional execution candidates omitted") + node["research_summary"] = "\n".join([node.get("research_summary") or "", *facts]).strip() + + +@contextmanager +def hold_research_completion_evidence( + *, registry_path: Path, runtime_root: Path, goal_id: str, todo: Mapping[str, Any], + state_text: str, actor_agent_id: str | None, +): + """Legacy writer transport; the same typed rule guards canonical commits.""" + from ...history import load_registry + from ...materials import find_registry_goal + from ...file_lock import exclusive_file_lock + from .result_log import build_explore_result_projection, explore_result_log_path, load_explore_result_events_strict + if not (todo.get("action_kind") == "joint_probe" or todo.get("explore_result_node_refs") + or todo.get("replan_obligation_id") or todo.get("capability_binding_ref")): + yield None + return + goal = find_registry_goal(load_registry(registry_path), goal_id) or {} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + policy = _research_result("explore.research.composition_policy", {"harness": harness}) + if not policy["enabled"] or todo.get("status") == "done" or todo.get("role") == "user": + yield None + return + todos = read_research_todo_history(runtime_root=runtime_root, goal=goal, state_text=state_text) + path = explore_result_log_path(runtime_root, goal_id) + with exclusive_file_lock(path, operation="research-legacy-completion"): + events = load_explore_result_events_strict(path, goal_id=goal_id) + projection = build_explore_result_projection(events, goal_id=goal_id) + frontier = build_research_composition_frontier(projection, + candidate_sources=[{"node_id": e["result_id"], "research_observation": e["research_observation"]} + for e in events if e.get("research_observation")], + harness=harness, todos=todos, agent_id=actor_agent_id or todo.get("claimed_by")) + guard = _research_result("explore.research.completion", {"frontier": frontier, + "todo": _research_todo_facts(todo), "actor_agent_id": actor_agent_id, "goal_id": goal_id}) + if guard["allowed"] is not True: + raise ValueError(str(guard["reason"])) + yield guard["evidence"] diff --git a/loopx/capabilities/explore/research_snapshot_host.py b/loopx/capabilities/explore/research_snapshot_host.py new file mode 100644 index 0000000000..113b9c6ac3 --- /dev/null +++ b/loopx/capabilities/explore/research_snapshot_host.py @@ -0,0 +1,45 @@ +"""Locked canonical graph transport for native completion. EOF releases locks. + +The existing Python Explore codec owns graph IO; TypeScript remains the owner +of eligibility, lineage and completion. No caller commands or task effects run. +""" +from __future__ import annotations + +import json +from pathlib import Path +import sys + +from ...file_lock import exclusive_file_lock +from .research_frontier import build_research_composition_frontier +from .result_log import build_explore_result_projection, explore_result_log_path, load_explore_result_events_strict + + +def main() -> None: + request = json.loads(sys.stdin.readline()) + root, goal = Path(request["runtime_root"]), request["goal_id"] + if not root.is_absolute(): + raise ValueError("absolute source runtime required") + path = explore_result_log_path(root, goal) + with exclusive_file_lock(path, operation="research-native-completion"): + events = load_explore_result_events_strict(path, goal_id=goal) + projection = build_explore_result_projection(events, goal_id=goal) + frontier = build_research_composition_frontier( + projection, candidate_sources=[ + {"node_id": event["result_id"], "research_observation": event["research_observation"]} + for event in events if event.get("research_observation") + ], harness=request["harness"], todos=request["todos"], agent_id=request["agent_id"], + ) + print(json.dumps({"schema_version": "research_graph_snapshot_v0", "frontier": frontier}, + ensure_ascii=False, allow_nan=False), flush=True) + # Keep the exact graph locked until the native transaction returns. + # Parent death closes this pipe and releases the OS lock automatically. + if sys.stdin.readline() != "release\n": + return + + +if __name__ == "__main__": + try: + main() + except (ValueError, OSError) as exc: + print(json.dumps({"schema_version": "research_graph_snapshot_v0", "error": type(exc).__name__}), flush=True) + raise SystemExit(1) from None diff --git a/loopx/chat_goal_configuration_api.py b/loopx/chat_goal_configuration_api.py index 5f3f2c6f79..9e76958ce0 100644 --- a/loopx/chat_goal_configuration_api.py +++ b/loopx/chat_goal_configuration_api.py @@ -112,12 +112,18 @@ def _peer_task_coordination_options(config: Mapping[str, Any]) -> dict[str, Any] def _explore_harness_options(config: Mapping[str, Any]) -> dict[str, Any]: profile = str(config.get("profile") or "").strip() or None + mode = config.get("composition_mode", "disabled") + scope = config.get("composition_scope_id") + if not isinstance(mode, str) or scope is not None and not isinstance(scope, str): + raise TypeError("Explore composition mode and scope must be strings") return { "explore_harness_enabled": _boolean_configuration( "explore_harness", config, "enabled" ), "explore_harness_profile": profile, "clear_explore_harness_profile": profile is None, + "explore_composition_mode": mode or "disabled", + "explore_composition_scope_id": scope or "", } @@ -212,7 +218,7 @@ def _goal_capability_options( }, "peer_task_coordination": {"coordinator_agent_id"}, "explore_graph": {"enabled"}, - "explore_harness": {"enabled", "profile"}, + "explore_harness": {"enabled", "profile", "composition_mode", "composition_scope_id"}, "pull_request_review": {"wait_for_ci", "review_priority"}, "change_quality_qualification": {"enabled", "safe_fix", "strict_receipt"}, "progress_review": {"mode", "signal", "drift_threshold", "contract_revision"}, diff --git a/loopx/cli_commands/explore.py b/loopx/cli_commands/explore.py index 8dfc604ab9..133cae8239 100644 --- a/loopx/cli_commands/explore.py +++ b/loopx/cli_commands/explore.py @@ -2,8 +2,9 @@ import argparse import json +from functools import partial from pathlib import Path -from typing import Callable +from typing import Any, Callable from ..capabilities.explore.result_log import ( DEFAULT_FINDING_LIMIT, @@ -133,6 +134,7 @@ def register_explore_commands( add_subcommand_format(summary) summary.add_argument("--goal-id", required=True) _add_projection_limit_args(summary) + summary.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") observe = sub.add_parser("observe", help="Record typed research evidence on an existing Explore node; grants no execution or closure authority.") add_subcommand_format(observe) @@ -147,6 +149,7 @@ def register_explore_commands( add_subcommand_format(presentation) presentation.add_argument("--goal-id", required=True) _add_projection_limit_args(presentation) + presentation.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") source_reconcile = sub.add_parser( "source-history-reconcile", @@ -172,6 +175,7 @@ def register_explore_commands( add_subcommand_format(graph) graph.add_argument("--goal-id", required=True) _add_projection_limit_args(graph) + graph.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") graph.add_argument( "--graph-format", choices=["mermaid", "json"], @@ -244,6 +248,8 @@ def _projection_for( *, runtime_root: Path, finding_limit_override: int | None = None, + registry: dict[str, Any] | None = None, + source_registry: Path | None = None, ) -> dict[str, object]: log_path = explore_result_log_path(runtime_root, args.goal_id) events = load_explore_result_events(log_path, goal_id=args.goal_id) @@ -261,6 +267,12 @@ def _projection_for( mermaid_node_limit=max(1, int(args.mermaid_node_limit)), ) projection["log_path"] = str(log_path) + if source_registry is not None: + registry = load_registry(source_registry) + if registry is not None: + from ..capabilities.explore.research_frontier import attach_research_execution_projection + attach_research_execution_projection(projection, events=events, registry=registry, + runtime_root=runtime_root, agent_id=getattr(args, "agent_id", None)) return projection @@ -328,6 +340,14 @@ def render_explore_markdown(payload: dict[str, object]) -> str: lines.append(f"- {key}: `{payload.get(key)}`") counts = payload.get("counts") research = payload.get("research_frontier") + live_research = payload.get("research_execution_frontier") + if isinstance(live_research, dict): + lines.append(f"- research execution ({live_research['agent_id']}): {live_research['pending_count']} pending, " + f"{live_research['scheduled_count']} scheduled, {live_research['observed_count']} observed, " + f"{live_research['ineligible_count']} ineligible, {live_research['dismissed_count']} dismissed, " + f"{live_research['deferred_count']} deferred") + for gap in live_research["gaps"]: + lines.append(f" - {gap['gap_id']}: {gap['status']}") if isinstance(research, dict): lines.append(f"- research composition (read-only): {research['pending_count']} pending, " f"{research['observed_count']} observed, {research['ineligible_count']} ineligible") @@ -483,6 +503,10 @@ def handle_explore_command( registry=registry, ) runtime_root = Path(str(source_runtime_route["source_runtime_root"])) + source_registry = Path(str(source_runtime_route["source_registry"])) if source_runtime_route else registry_path + same_source = source_registry.resolve() == registry_path.resolve() + projection_for = partial(_projection_for, registry=registry if same_source else None, + source_registry=None if same_source else source_registry) config_path = ( Path(args.config_path).expanduser() if getattr(args, "config_path", None) @@ -546,11 +570,13 @@ def handle_explore_command( payload = append_research_observation( explore_result_log_path(runtime_root, args.goal_id), goal_id=args.goal_id, observation=observation, agent_id=args.agent_id, + registry_path=Path(str(source_runtime_route["source_registry"])) if source_runtime_route else registry_path, + runtime_root=runtime_root, ) elif args.explore_command == "summary": - payload = _projection_for(args, runtime_root=runtime_root) + payload = projection_for(args, runtime_root=runtime_root) elif args.explore_command == "presentation": - projection = _projection_for( + projection = projection_for( args, runtime_root=runtime_root, finding_limit_override=-1, @@ -573,7 +599,7 @@ def handle_explore_command( execute=bool(args.execute), ) elif args.explore_command == "graph": - projection = _projection_for(args, runtime_root=runtime_root) + projection = projection_for(args, runtime_root=runtime_root) graph_view = build_explore_graph_view( projection.get("nodes") or [], projection.get("edges") or [], @@ -637,7 +663,7 @@ def handle_explore_command( resource_usage=resource_usage, ) else: - projection = _projection_for(args, runtime_root=runtime_root) + projection = projection_for(args, runtime_root=runtime_root) todo_payload = list_goal_todos( registry_path=registry_path, goal_id=args.goal_id, @@ -688,7 +714,7 @@ def handle_explore_command( resource_usage=resource_usage, ) else: - projection = _projection_for(args, runtime_root=runtime_root) + projection = projection_for(args, runtime_root=runtime_root) todo_payload = list_goal_todos( registry_path=registry_path, goal_id=args.goal_id, @@ -730,7 +756,7 @@ def handle_explore_command( args, config_path=config_path, runtime_root=runtime_root, - projection_for=_projection_for, + projection_for=projection_for, ) else: raise ValueError(f"unknown explore command: {args.explore_command}") diff --git a/loopx/cli_commands/explore_feishu_commands.py b/loopx/cli_commands/explore_feishu_commands.py index c8e5b26208..a45acbbef5 100644 --- a/loopx/cli_commands/explore_feishu_commands.py +++ b/loopx/cli_commands/explore_feishu_commands.py @@ -141,6 +141,7 @@ def register_explore_feishu_commands( add_subcommand_format(sync) add_config_path_arg(sync) sync.add_argument("--goal-id", required=True) + sync.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") add_projection_limit_args(sync) sync.add_argument("--base-token") for table_key in EXPLORE_TABLE_KEYS: @@ -166,6 +167,7 @@ def register_explore_feishu_commands( add_subcommand_format(card) add_config_path_arg(card) card.add_argument("--goal-id", required=True) + card.add_argument("--agent-id", help="Registered agent whose live research lineage is displayed.") add_projection_limit_args(card) card.add_argument("--title") card.add_argument("--template", default="blue") diff --git a/loopx/cli_commands/project_lifecycle_inputs.py b/loopx/cli_commands/project_lifecycle_inputs.py index fd13b1413f..d7ea5aa317 100644 --- a/loopx/cli_commands/project_lifecycle_inputs.py +++ b/loopx/cli_commands/project_lifecycle_inputs.py @@ -63,6 +63,7 @@ def inline_progress_observation( args: argparse.Namespace, ) -> dict[str, object] | None: fields = { + "work_item_id": getattr(args, "progress_work_item_id", None), "surface_id": getattr(args, "progress_surface_id", None), "hypothesis_id": getattr(args, "progress_hypothesis_id", None), "probe_kind": getattr(args, "progress_probe_kind", None), diff --git a/loopx/cli_commands/project_lifecycle_refresh_state.py b/loopx/cli_commands/project_lifecycle_refresh_state.py index b316dc034e..1dbfd7bfbc 100644 --- a/loopx/cli_commands/project_lifecycle_refresh_state.py +++ b/loopx/cli_commands/project_lifecycle_refresh_state.py @@ -195,6 +195,7 @@ def register_refresh_state_command( ), ) refresh_state_parser.add_argument("--progress-surface-id") + refresh_state_parser.add_argument("--progress-work-item-id", help="Explicit typed observation source; settlement keeps its own Todo or obligation binding.") refresh_state_parser.add_argument("--progress-hypothesis-id") refresh_state_parser.add_argument("--progress-probe-kind") refresh_state_parser.add_argument("--progress-blocker-id") diff --git a/loopx/cli_commands/registry_admin.py b/loopx/cli_commands/registry_admin.py index 5a293273ec..4fc2af0864 100644 --- a/loopx/cli_commands/registry_admin.py +++ b/loopx/cli_commands/registry_admin.py @@ -484,6 +484,8 @@ def handle_registry_admin_command( explore_harness_enabled=args.explore_harness_enabled, explore_harness_profile=args.explore_harness_profile, clear_explore_harness_profile=bool(args.clear_explore_harness_profile), + explore_composition_mode=args.explore_composition_mode, + explore_composition_scope_id=args.explore_composition_scope_id, explore_graph_enabled=args.explore_graph_enabled, lark_kanban_heartbeat_sync=args.lark_kanban_heartbeat_sync, registered_agents=args.registered_agents, diff --git a/loopx/cli_commands/registry_admin_configure.py b/loopx/cli_commands/registry_admin_configure.py index 454dc62f12..4f55ed18f5 100644 --- a/loopx/cli_commands/registry_admin_configure.py +++ b/loopx/cli_commands/registry_admin_configure.py @@ -242,6 +242,14 @@ def register_configure_goal_command(subparsers: argparse._SubParsersAction) -> N action="store_true", help="Remove the goal-pinned explore profile while preserving the opt-in bit.", ) + configure_goal_parser.add_argument( + "--explore-composition-mode", choices=["disabled", "explicit_only"], + help="Opt in to typed composition obligations for this Goal; disabled preserves existing planning behavior.", + ) + configure_goal_parser.add_argument( + "--explore-composition-scope-id", + help="Opaque research coverage scope required by explicit_only composition; grants no execution authority.", + ) configure_goal_parser.add_argument( "--registered-agent", dest="registered_agents", diff --git a/loopx/cli_commands/status.py b/loopx/cli_commands/status.py index 89d9d6e72f..daa6e4167d 100644 --- a/loopx/cli_commands/status.py +++ b/loopx/cli_commands/status.py @@ -18,6 +18,7 @@ compact_agent_lane_status_payload_for_display, ) from ..control_plane.todos.contract import normalize_todo_claimed_by +from ..control_plane.quota.goal_boundary import registry_goal_by_id from ..control_plane.todos.quota_summary import ( compact_agent_lane_todos_for_status_display, ) @@ -564,7 +565,9 @@ def _sync_agent_replan_obligation_from_guard( """Keep agent-scoped status guidance identical to the quota decision.""" targets = (item, project_asset) - if not any( + obligation = guard.get("autonomous_replan_obligation") + capability_obligation = isinstance(obligation, dict) and bool(obligation.get("capability_guard")) + if not capability_obligation and not any( isinstance(target, dict) and ( "autonomous_replan_obligation" in target @@ -576,7 +579,6 @@ def _sync_agent_replan_obligation_from_guard( for target in targets ): return - obligation = guard.get("autonomous_replan_obligation") for target in targets: if not isinstance(target, dict): continue @@ -665,22 +667,38 @@ def attach_agent_lane_next_actions( "agent_member", "agent_interaction_summary", "agent_reward_memory", + "bounded_research_frontier", ), 0, ) current_agent_next_action: dict[str, Any] | None = None + goals = registry_goal_by_id(payload) for item in items: if not isinstance(item, dict): continue goal_id = str(item.get("goal_id") or "").strip() if not goal_id: continue + decision_payload = payload + frontier = None + goal = goals.get(goal_id) or {} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + if harness.get("enabled") is True and harness.get("composition_mode") == "explicit_only": + from ..capabilities.explore.composition_frontier import project_live_explore_composition_frontier + frontier = project_live_explore_composition_frontier( + runtime_root=Path(str(payload["runtime_root"])), goal_id=goal_id, + agent_id=safe_agent_id, status_payload=payload) + decision_payload = {**payload, "bounded_research_frontier": frontier} try: guard = build_quota_should_run( - payload, + decision_payload, goal_id=goal_id, agent_id=safe_agent_id, ) + selected = guard.get("replan_action_packet") or {} + obligation = (frontier or {}).get("obligation") or {} + if selected.get("capability_guard") and selected.get("obligation_id") == obligation.get("obligation_id"): + guard["autonomous_replan_obligation"] = obligation except Exception: continue next_action = guard.get("agent_lane_next_action") @@ -715,6 +733,7 @@ def attach_agent_lane_next_actions( "agent_member": agent_member, "agent_interaction_summary": interaction_summary, "agent_reward_memory": reward_memory_projection, + "bounded_research_frontier": guard.get("bounded_research_frontier"), }, ) for field in attached_fields: diff --git a/loopx/cli_commands/todo.py b/loopx/cli_commands/todo.py index f08c84cc1d..2594b2bfc2 100644 --- a/loopx/cli_commands/todo.py +++ b/loopx/cli_commands/todo.py @@ -209,7 +209,11 @@ def _validated_replan_successor_obligation( ), None, ) + from ..capabilities.explore.research_frontier import prepare_research_replan_evidence + capability_evidence = prepare_research_replan_evidence(runtime_root=runtime_root, goal_id=args.goal_id, + agent_id=args.claimed_by, registry_goal=registry_goal, state_text=state_text) obligation, _ = qualify_replan_writeback( + capability_evidence=capability_evidence, todo_fields=todo_fields, newest_first_runs=newest_first_runs, state_text=state_text, @@ -228,6 +232,17 @@ def _validated_replan_successor_obligation( "--replan-obligation-id does not match the current open obligation: " f"expected {current}" ) + if (obligation or {}).get("capability_guard"): + from ..capabilities.explore.research_evidence import _research_result + if capability_evidence is None: + raise ValueError("selected capability successor owner is not available") + _research_result("explore.research.composition_successor", { + "frontier": capability_evidence.frontier, "agent_id": args.claimed_by, "obligation_id": requested, + "todo": {"replan_obligation_id": requested, "claimed_by": args.claimed_by, + "action_kind": args.action_kind, "task_class": args.task_class, + "explore_result_node_refs": args.explore_result_node_refs, + "target_key": args.monitor_target_key, "status": args.status or "open", "resume_when": args.resume_when}, + }) return current diff --git a/loopx/cli_commands/turn.py b/loopx/cli_commands/turn.py index 496d85c700..fa1e458720 100644 --- a/loopx/cli_commands/turn.py +++ b/loopx/cli_commands/turn.py @@ -1043,6 +1043,7 @@ def on_managed_start_admitted() -> None: settlement_identity, semantic_replan_guard_scoped=replan_guard_scoped, semantic_replan_obligation_id=replan_obligation_id, + semantic_replan_capability_guard=(stable_envelope.get("replan_action_packet") or {}).get("capability_guard"), ) managed_cadence = managed_cadence_start( diff --git a/loopx/configuration_catalog.py b/loopx/configuration_catalog.py index d4d9d1f4d2..f339bb7aee 100644 --- a/loopx/configuration_catalog.py +++ b/loopx/configuration_catalog.py @@ -514,13 +514,15 @@ def build_goal_configuration_catalog( "current": { "enabled": harness.get("enabled") is True, "profile": harness.get("profile"), + "composition_mode": harness.get("composition_mode", "disabled"), + "composition_scope_id": harness.get("composition_scope_id"), }, "profiles": list(explore_harness_profiles), "consider_when": ( "The goal benefits from comparing alternative branches with explicit " "evaluation criteria and guardrails." ), - "effect": "Enables read-only Explore branch and worker-lane planning.", + "effect": "Enables read-only Explore planning. Explicit composition replans require an exact experiment successor or typed result; task completion remains separate.", "does_not": [ "enable Explore Graph", "launch workers, claim todos, acquire leases, mutate state, or spend quota", diff --git a/loopx/configure_goal.py b/loopx/configure_goal.py index 248448d9ee..03686c4fb3 100644 --- a/loopx/configure_goal.py +++ b/loopx/configure_goal.py @@ -467,6 +467,8 @@ def configure_goal( explore_harness_enabled: bool | None = None, explore_harness_profile: str | None = None, clear_explore_harness_profile: bool = False, + explore_composition_mode: str | None = None, + explore_composition_scope_id: str | None = None, explore_graph_enabled: bool | None = None, registered_agents: list[str] | None = None, clear_registered_agents: bool = False, @@ -978,6 +980,8 @@ def configure_goal( or explore_harness_enabled is not None or explore_harness_profile is not None or clear_explore_harness_profile + or explore_composition_mode is not None + or explore_composition_scope_id is not None ): spawn_policy = ( goal.get("spawn_policy") @@ -1003,6 +1007,8 @@ def configure_goal( explore_harness_enabled is not None or explore_harness_profile is not None or clear_explore_harness_profile + or explore_composition_mode is not None + or explore_composition_scope_id is not None ): explore_harness = ( spawn_policy.get("explore_harness") @@ -1015,6 +1021,15 @@ def configure_goal( explore_harness.pop("profile", None) elif explore_harness_profile is not None: explore_harness["profile"] = explore_harness_profile + if explore_composition_mode is not None: + explore_harness["composition_mode"] = explore_composition_mode + if explore_composition_scope_id is not None: + if explore_composition_scope_id: + explore_harness["composition_scope_id"] = explore_composition_scope_id + else: + explore_harness.pop("composition_scope_id", None) + from .orchestration import compact_explore_harness_policy + compact_explore_harness_policy(explore_harness) if explore_harness: spawn_policy["explore_harness"] = explore_harness else: diff --git a/loopx/control_plane/capabilities/explore_research.ts b/loopx/control_plane/capabilities/explore_research.ts index 40722c1aec..1114ddd393 100644 --- a/loopx/control_plane/capabilities/explore_research.ts +++ b/loopx/control_plane/capabilities/explore_research.ts @@ -45,10 +45,11 @@ function digest(value: unknown): string { .sort(([a], [b]) => compare(a, b)).map(([key, child]) => [key, stable(child)])) : item; return createHash("sha256").update(JSON.stringify(stable(value))).digest("hex").slice(0, 16); } +export {id as researchIdentifier, digest as researchDigest}; export function normalizeResearchObservation(params: JsonObject): JsonObject { const raw = object(params.observation, "research observation", - ["schema_version", "explore_node_id", "progress", "closure_basis", "composition_candidates", "input_observations", "fingerprint"]); + ["schema_version", "explore_node_id", "progress", "closure_basis", "composition_candidates", "input_observations", "execution_lineage", "composition_resolution", "fingerprint"]); requireStringLiteral(raw.schema_version, [OBSERVATION], "research observation schema"); const node = id(raw.explore_node_id, "explore_node_id"); // The Python transport composes its existing generic progress codec. This @@ -109,8 +110,47 @@ export function normalizeResearchObservation(params: JsonObject): JsonObject { if (new Set(lineage.map(input => input.node_id)).size !== lineage.length || lineage.some(input => input.node_id === node)) { throw new EffectRuntimeRequestError("input observations must have distinct non-self node identities"); } + let execution: JsonObject | null = null; + if (raw.execution_lineage !== undefined) { + const value = object(raw.execution_lineage, "execution_lineage", ["schema_version", "goal_id", "gap_id", "replan_obligation_id", "successor_todo_id", "agent_id"]); + requireStringLiteral(value.schema_version, ["research_execution_lineage_v0"], "execution lineage schema"); + execution = {schema_version: value.schema_version}; + for (const field of ["goal_id", "gap_id", "replan_obligation_id", "successor_todo_id", "agent_id"]) { + execution[field] = id(value[field], `execution_lineage.${field}`); + } + if (!/^research-composition-[a-f0-9]{16}$/.test(String(execution.gap_id)) + || !/^replan-[a-f0-9]{16}$/.test(String(execution.replan_obligation_id)) + || !/^todo_[A-Za-z0-9_-]+$/.test(String(execution.successor_todo_id)) + || progress.work_item_id !== execution.successor_todo_id || lineage.length !== 2) { + throw new EffectRuntimeRequestError("execution lineage requires exact gap, obligation, Todo and binary input identities"); + } + } + let resolution: JsonObject | null = null; + if (raw.composition_resolution !== undefined) { + const value = object(raw.composition_resolution, "composition_resolution", + ["schema_version", "disposition", "basis", "evidence_ids"]); + requireStringLiteral(value.schema_version, ["research_composition_resolution_v0"], "composition resolution schema"); + const disposition = requireStringLiteral(value.disposition, ["dismissed", "deferred"], "resolution disposition"); + const refs = ids(value.evidence_ids, "resolution evidence_ids"); + if (execution === null || !refs.length || refs.some(ref => !evidence.includes(ref))) { + throw new EffectRuntimeRequestError("composition resolution requires execution lineage and attributable evidence"); + } + resolution = {schema_version: value.schema_version, disposition, evidence_ids: refs}; + if (disposition === "dismissed") { + if (result !== "no_followup" || closure?.disposition !== "no_followup") { + throw new EffectRuntimeRequestError("candidate dismissal requires coverage-backed no_followup and its closure basis"); + } + resolution.basis = requireStringLiteral(value.basis, ["duplicate", "invalid", "unsafe", "outside_scope"], "dismissal basis"); + } else if (result !== "blocked" || value.basis !== undefined + || progress.coverage_complete === true || closure?.disposition === "no_followup" || closure?.disposition === "exhausted" + || !/^todo_[A-Za-z0-9_-]+$/.test(String(progress.blocker_id))) { + throw new EffectRuntimeRequestError("temporary composition deferral requires blocked progress with a canonical blocker Todo id"); + } + } const canonical: JsonObject = {schema_version: OBSERVATION, explore_node_id: node, progress, ...(lineage.length ? {input_observations: lineage} : {}), + ...(execution ? {execution_lineage: execution} : {}), + ...(resolution ? {composition_resolution: resolution} : {}), ...(closure ? {closure_basis: closure} : {}), composition_candidates: candidates}; return {...canonical, fingerprint: digest(canonical)}; } @@ -124,7 +164,7 @@ function observationFor(node: JsonObject): JsonObject | null { function eligible(node: JsonObject | undefined): boolean { if (!node || !["resolved", "dead_end"].includes(String(node.status))) return false; const observation = observationFor(node); - return !!observation && TERMINAL.has(String((observation.progress as JsonObject).result_class)); + return !!observation && !observation.composition_resolution && TERMINAL.has(String((observation.progress as JsonObject).result_class)); } function evidenceFor(node: JsonObject): Set { const observation = observationFor(node); @@ -164,7 +204,8 @@ export function validateResearchAttribution(params: JsonObject): JsonObject { return observation; } -export function projectResearchFrontier(params: JsonObject): JsonObject { +/** Full internal candidate set; presentation compaction cannot hide a write gate. */ +export function researchCompositionGaps(params: JsonObject): JsonObject[] { const goal = id(params.goal_id, "goal_id"); const nodeRows = rows(params.nodes, "nodes"); const nodes = new Map(nodeRows.map(node => [String(node.node_id), node])); @@ -206,23 +247,35 @@ export function projectResearchFrontier(params: JsonObject): JsonObject { return sets.every(set => refs.some(ref => set.has(ref))) && refs.every(ref => sets.some(set => set.has(ref))); }); const experiments = experimentsByInputs.get(JSON.stringify(candidate.inputs)) ?? []; - const observed = allEligible && attributable && experiments.some(experiment => { - if (!eligible(experiment)) return false; - const lineage = observationFor(experiment)?.input_observations as JsonObject[] ?? []; + const terminal = allEligible && attributable ? experiments.filter(experiment => { + const outcome = observationFor(experiment); + if (!["resolved", "dead_end"].includes(String(experiment.status)) + || !outcome || !TERMINAL.has(String((outcome.progress as JsonObject).result_class))) return false; + const lineage = outcome.input_observations as JsonObject[] ?? []; return JSON.stringify(lineage.map(input => input.node_id)) === JSON.stringify(candidate.inputs) && lineage.every(input => observationFor(nodes.get(String(input.node_id))!)?.fingerprint === input.fingerprint); - }); + }) : []; + const observed = terminal.some(experiment => !observationFor(experiment)?.composition_resolution); + const dismissed = terminal.some(experiment => experiment.status === "dead_end" + && (observationFor(experiment)?.composition_resolution as JsonObject)?.disposition === "dismissed"); const active = experiments.filter(node => ["open", "exploring"].includes(String(node.status))); gaps.push({gap_id: `research-composition-${identity}`, input_node_ids: candidate.inputs, input_observations: inputs.filter((node): node is JsonObject => !!node).map(node => ({node_id: node.node_id, fingerprint: observationFor(node)?.fingerprint ?? null})), - state: !allEligible || !attributable ? "ineligible" : observed ? "observed" : "pending", + state: !allEligible || !attributable ? "ineligible" : observed ? "observed" : dismissed ? "dismissed" : "pending", reason: !allEligible ? "terminal_input_observation_required" : !attributable ? "input_evidence_invalidated" - : observed ? "typed_experiment_outcome" : "joint_experiment_result_required", + : observed ? "typed_experiment_outcome" : dismissed ? "typed_candidate_dismissal" : "joint_experiment_result_required", interaction_kinds: [...new Set(candidate.claims.map(claim => String(claim.interaction_kind)))].sort(), experiment_node_ids: experiments.map(node => String(node.node_id)).sort(), active_experiment_node_ids: active.map(node => String(node.node_id)).sort()}); } + return gaps; +} + +export function projectResearchFrontier(params: JsonObject): JsonObject { + const goal = id(params.goal_id, "goal_id"); + const nodeRows = rows(params.nodes, "nodes"); + const gaps = researchCompositionGaps(params); const count = (state: string) => gaps.filter(gap => gap.state === state).length; const gapsByInput = new Map(); for (const gap of gaps) for (const input of gap.input_node_ids as string[]) { @@ -239,12 +292,14 @@ export function projectResearchFrontier(params: JsonObject): JsonObject { progress ? `Research: ${String(progress.result_class).replaceAll("_", " ")}` : "Research: observation invalidated", ...(progress?.coverage_scope_id ? [`coverage ${progress.coverage_scope_id}`] : []), ...(basis ? [`closure ${basis.disposition}`] : []), + ...(observation?.composition_resolution ? [`composition resolution ${(observation.composition_resolution as JsonObject).disposition}`] : []), ...(related.length ? [`composition ${related.filter(gap => gap.state === "pending").length} pending, ${related.filter(gap => gap.state === "ineligible").length} ineligible, ${related.filter(gap => gap.state === "observed").length} observed`] : []), ].join("; ")}; }); return {schema_version: "research_frontier_projection_v0", goal_id: goal, mode: "read_only_shadow", candidate_count: gaps.length, pending_count: count("pending"), observed_count: count("observed"), ineligible_count: count("ineligible"), projected_count: Math.min(gaps.length, MAX_RESEARCH_GAPS), + ...(nodeRows.some(node => observationFor(node)?.composition_resolution) ? {dismissed_count: count("dismissed")} : {}), omitted_count: Math.max(0, gaps.length - MAX_RESEARCH_GAPS), gaps: gaps.slice(0, MAX_RESEARCH_GAPS), grants_execution_authority: false, node_summaries: nodeSummaries}; } diff --git a/loopx/control_plane/capabilities/explore_research_execution.ts b/loopx/control_plane/capabilities/explore_research_execution.ts new file mode 100644 index 0000000000..390584d30a --- /dev/null +++ b/loopx/control_plane/capabilities/explore_research_execution.ts @@ -0,0 +1,321 @@ +/** Explore execution attribution. This records evidence, never work authority. */ +import type {JsonObject} from "../effect_program.ts"; +import {EffectRuntimeRequestError} from "../effect_runtime_errors.ts"; +import {requireJsonObject} from "../runtime_decode.ts"; +import {normalizeResearchObservation, researchCompositionGaps, researchIdentifier, researchDigest, + validateResearchAttribution} from "./explore_research.ts"; +import {evaluateTodoResumeConditions, TODO_RESUME_EVALUATION_REQUEST_SCHEMA_VERSION, + resumeConditionHasKnownPendingTarget} from "../todos/resume_condition.ts"; + +const object = (value: unknown): JsonObject => value && typeof value === "object" && !Array.isArray(value) ? value as JsonObject : {}; +const array = (value: unknown): JsonObject[] => Array.isArray(value) ? value.map(row => requireJsonObject(row, "research row")) : []; +const same = (left: unknown, right: unknown) => JSON.stringify(left) === JSON.stringify(right); +type CompositionFrontierState = "disabled" | "pending" | "scheduled" | "observed" | "dismissed" | "deferred" | "ineligible" | "empty"; + +export function normalizeResearchCompositionPolicy(params: JsonObject): JsonObject { + const harness = requireJsonObject(params.harness, "Explore harness"); + const mode = harness.composition_mode ?? "disabled"; + if (!["disabled", "explicit_only"].includes(String(mode))) { + throw new EffectRuntimeRequestError("composition_mode must be disabled or explicit_only"); + } + const scope = harness.composition_scope_id == null || harness.composition_scope_id === "" ? null + : researchIdentifier(harness.composition_scope_id, "composition_scope_id"); + if (mode === "explicit_only" && scope === null) { + throw new EffectRuntimeRequestError("explicit_only composition requires composition_scope_id"); + } + return {schema_version: "research_composition_policy_v0", mode, coverage_scope_id: scope, + enabled: harness.enabled === true && mode === "explicit_only"}; +} + +function boundTodo(todo: JsonObject, gap: JsonObject, obligationId: unknown, agent: unknown, + retainedHistory = false): boolean { + const refs = todo.explore_result_node_refs; + const completed = retainedHistory && todo.status === "done"; + return (todo.claimed_by === agent || completed && todo.claimed_by == null) && todo.replan_obligation_id === obligationId + && (!todo.role || todo.role === "agent") + && todo.task_class === "advancement_task" && todo.action_kind === "joint_probe" + && (completed || todo.archive_state !== "archive") && Array.isArray(refs) && refs.length === 1 + && (gap.experiment_node_ids as string[]).includes(String(refs[0])) + && (!todo.target_key || todo.target_key === refs[0]) + && !(Array.isArray(todo.excluded_agents) && todo.excluded_agents.includes(agent)); +} + +/** Internal unbounded facts for obligation identity transport. Never an extra + * public list: CLI/status receive only the compact frontier below. */ +export function researchCompositionFacts(params: JsonObject): JsonObject { + const policy = normalizeResearchCompositionPolicy(params); + if (!policy.enabled) return {policy, gaps: []}; + const gaps = researchCompositionGaps(params).map(gap => ({...gap, + frontier_revision: `research-composition-v0:${researchDigest([policy, gap.gap_id, gap.input_observations])}`})); + return {policy, gaps}; +} + +/** Join current evidence and canonical Todo facts. Only exact lineage can + * schedule or observe a gap; historical diagnostic observations stay cold. */ +export function projectResearchComposition(params: JsonObject): JsonObject { + const {policy, gaps: raw} = researchCompositionFacts(params); + const gaps = raw as JsonObject[], todos = array(params.todos), bindings = array(params.bindings); + const nodes = array(params.nodes); + const nodeById = new Map(nodes.map(node => [node.node_id, node])); + const todoById = new Map(todos.map(todo => [todo.todo_id, todo])); + if (todoById.size !== todos.length || nodeById.size !== nodes.length) { + throw new EffectRuntimeRequestError("canonical research node and Todo identities must be unique"); + } + const todosByObligation = new Map(); + for (const todo of todos) { + // A terminal record may clear its claim. Its accepted observation retains + // the actor; this historical join cannot grant runnable work authority. + if (todo.claimed_by !== params.agent_id && !(todo.status === "done" && todo.claimed_by == null)) continue; + const owned = todosByObligation.get(todo.replan_obligation_id) ?? []; + owned.push(todo); todosByObligation.set(todo.replan_obligation_id, owned); + } + const bindingByGap = new Map(bindings.map(binding => [binding.gap_id, requireJsonObject(binding.obligation, "composition obligation")])); + const resumeByTodo = new Map(array((policy as JsonObject).enabled ? evaluateTodoResumeConditions({ + schema_version: TODO_RESUME_EVALUATION_REQUEST_SCHEMA_VERSION, + items: todos, source_items: todos, kinds: ["todo_done"], rollout_events: [], + }).conditions : []).map(row => [row.todo_id, object(row.condition)])); + const projected: JsonObject[] = []; + for (const gap of gaps) { + const obligation = bindingByGap.get(gap.gap_id); + if (!obligation || !/^replan-[a-f0-9]{16}$/.test(String(obligation.obligation_id))) { + throw new EffectRuntimeRequestError("composition gap requires its common obligation identity"); + } + const matched = (todosByObligation.get(obligation.obligation_id) ?? []) + .filter(todo => boundTodo(todo, gap, obligation.obligation_id, params.agent_id, true)); + const results = (gap.experiment_node_ids as string[]).map(id => nodeById.get(id)) + .flatMap(node => { + if (!node?.research_observation || gap.state === "ineligible") return []; + const observation = normalizeResearchObservation({observation: node.research_observation}); + const lineage = object(observation.execution_lineage), progress = object(observation.progress); + const todo = todoById.get(lineage.successor_todo_id); + const resolution = object(observation.composition_resolution); + const attributable = !!todo && matched.includes(todo) + && (["open", "claimed", "done"].includes(String(todo.status)) + || resolution.disposition === "deferred" && ["blocked", "deferred"].includes(String(todo.status))) + && lineage.goal_id === params.goal_id && lineage.agent_id === params.agent_id && lineage.gap_id === gap.gap_id + && lineage.replan_obligation_id === obligation.obligation_id + && progress.work_item_id === todo.todo_id && progress.coverage_scope_id === (policy as JsonObject).coverage_scope_id + && same(observation.input_observations, gap.input_observations); + return attributable ? [{node, observation, progress, todo, resolution}] : []; + }); + const observed = results.find(result => ["resolved", "dead_end"].includes(String(result.node.status)) + && !result.resolution.disposition && ["exploration_exhausted", "no_followup"].includes(String(result.progress.result_class))); + const dismissed = results.find(result => result.node.status === "dead_end" && result.resolution.disposition === "dismissed"); + const deferred = results.find(result => { + if (result.node.status !== "blocked" || result.resolution.disposition !== "deferred" + || !["blocked", "deferred"].includes(String(result.todo.status))) return false; + const blocker = todoById.get(result.progress.blocker_id), condition = resumeByTodo.get(result.todo.todo_id); + return !!blocker && blocker.role === "agent" && blocker.task_class === "blocker" + && blocker.claimed_by === params.agent_id && blocker.archive_state !== "archive" + && ["open", "deferred"].includes(String(blocker.status)) && blocker.unblocks_todo_id === result.todo.todo_id + && !!condition && condition.kind === "todo_done" && condition.target_todo_id === blocker.todo_id + && resumeConditionHasKnownPendingTarget(condition, result.todo); + }); + const successor = gap.state !== "ineligible" ? matched.find(todo => todo.actionable_open === true + && boundTodo(todo, gap, obligation.obligation_id, params.agent_id) && ["open", "claimed"].includes(String(todo.status)) + && ["open", "exploring"].includes(String(nodeById.get((todo.explore_result_node_refs as string[])[0])?.status))) : undefined; + const experiment = successor ? (successor.explore_result_node_refs as string[])[0] + : (gap.active_experiment_node_ids as string[])[0] ?? null; + projected.push({...gap, status: gap.state === "ineligible" ? "ineligible" : observed ? "observed" : dismissed ? "dismissed" + : successor ? "scheduled" : deferred ? "deferred" : "pending", + obligation_id: obligation.obligation_id, experiment_node_ref: experiment, + input_node_refs: gap.input_node_ids, required_outcome: "typed_joint_experiment_result", + successor_todo_id: successor?.todo_id ?? null, + observed_todo_id: observed?.todo?.todo_id ?? null, + observed_progress_fingerprint: observed?.progress.fingerprint ?? null, + dismissed_todo_id: dismissed?.todo.todo_id ?? null, + deferred_todo_id: deferred?.todo.todo_id ?? null, + blocker_todo_id: deferred?.progress.blocker_id ?? null, + resume_when: deferred?.todo.resume_when ?? null, + execution_results: results.map(result => ({todo_id: result.todo?.todo_id, experiment_node_id: result.node.node_id, + observation_fingerprint: result.observation.fingerprint, progress_fingerprint: result.progress.fingerprint, + progress: result.progress, + result_class: result.progress.result_class, blocker_id: result.progress.blocker_id ?? null, + disposition: result.resolution.disposition ?? null, + evidence_ids: result.progress.evidence_ids})), + successor_summary: experiment ? `Run the bounded joint experiment: ${experiment}` : "Create one binary experiment for this explicit input set, then bind a runnable Todo.", + successor_binding: experiment ? {action_kind: "joint_probe", task_domain: "research", target_key: experiment, + explore_result_node_refs: [experiment]} : null}); + } + projected.sort((a, b) => Number(a.status !== "pending") - Number(b.status !== "pending") + || String(a.gap_id).localeCompare(String(b.gap_id), "en")); + const selected = projected.find(gap => gap.status === "pending") ?? null; + const state: CompositionFrontierState = !(policy as JsonObject).enabled ? "disabled" : selected ? "pending" + : projected.some(gap => gap.status === "scheduled") ? "scheduled" + : projected.some(gap => gap.status === "deferred") ? "deferred" + : projected.some(gap => gap.status === "ineligible") ? "ineligible" + : projected.some(gap => gap.status === "observed") ? "observed" : projected.length ? "dismissed" : "empty"; + return {schema_version: "research_composition_frontier_v0", goal_id: params.goal_id, agent_id: params.agent_id, policy, + grants_execution_authority: false, + enabled: (policy as JsonObject).enabled, state, + candidate_count: gaps.length, pending_count: projected.filter(gap => gap.status === "pending").length, + scheduled_count: projected.filter(gap => gap.status === "scheduled").length, + observed_count: projected.filter(gap => gap.status === "observed").length, + dismissed_count: projected.filter(gap => gap.status === "dismissed").length, + deferred_count: projected.filter(gap => gap.status === "deferred").length, + ineligible_count: projected.filter(gap => gap.status === "ineligible").length, + omitted_count: Math.max(0, projected.length - 3), gaps: projected.slice(0, 3), selected_gap: selected, + // This internal lane is removed by the transport before public projection. + lineage_gaps: projected, + obligation_tasks: todos.filter(todo => todo.replan_obligation_id) + .map(todo => ({obligation_id: todo.replan_obligation_id, todo_id: todo.todo_id, + status: todo.status, archive_state: todo.archive_state ?? "active"}))}; +} + +function retirementContract(frontier: JsonObject, guard: JsonObject, obligation: unknown): JsonObject | null { + if (frontier.schema_version !== "research_composition_frontier_v0" + || guard.schema_version !== "semantic_replan_capability_guard_v0" || guard.capability_id !== "explore") return null; + const current = array(frontier.lineage_gaps).find(row => row.gap_id === guard.gap_id); + if (frontier.enabled === true && (!current || current.frontier_revision === guard.frontier_revision + && current.status !== "ineligible")) return null; + const blocking = array(frontier.obligation_tasks).filter(todo => todo.obligation_id === obligation + && todo.archive_state !== "archive" && ["open", "claimed"].includes(String(todo.status))); + const revision = `capability-evidence-v0:${researchDigest([frontier.policy, current ?? null])}`; + const blocker = `capability-invalidated-${researchDigest([guard, revision])}`; + return {schema_version: "capability_obligation_retirement_v0", disposition: "invalidated", + reason_code: frontier.enabled !== true ? "source_disabled" : current?.status === "ineligible" + ? "source_ineligible" : "source_revision_changed", + capability_id: "explore", obligation_id: obligation, original_guard: guard, current_revision: revision, + blocking_todo_ids: blocking.slice(0, 3).map(todo => todo.todo_id), blocking_todo_count: blocking.length, + progress_observation: {schema_version: "typed_progress_observation_v0", result_class: "blocked", + work_item_id: obligation, blocker_id: blocker, evidence_ids: [revision]}}; +} + +/** Qualify the exact research duty selected at admission. Generic progress or + * a vision/read/ACK cannot impersonate the canonical experiment observation. */ +export function qualifyResearchCompositionWriteback(params: JsonObject): JsonObject { + const frontier = requireJsonObject(params.frontier, "live research frontier"); + const guard = requireJsonObject(params.capability_guard, "selected capability guard"); + const gap = array(frontier.lineage_gaps).find(row => row.gap_id === guard.gap_id + && row.frontier_revision === guard.frontier_revision && row.obligation_id === params.obligation_id); + const observation = object(params.progress_observation); + const progressFingerprint = observation.fingerprint; + const repeated = Array.isArray(params.claimed_progress_fingerprints) && params.claimed_progress_fingerprints.includes(progressFingerprint); + let outcome: string | null = null, capabilityOutcome: string | null = null; + let result: JsonObject | undefined; + const retirement = gap && gap.status !== "ineligible" ? null : retirementContract(frontier, guard, params.obligation_id); + let retired = false; + if (retirement && retirement.blocking_todo_count === 0 && typeof progressFingerprint === "string") { + const claimed = {...observation}; delete claimed.fingerprint; + if (researchDigest(claimed) === researchDigest(retirement.progress_observation)) { + outcome = "capability_duty_retired"; capabilityOutcome = "composition_duty_invalidated"; retired = true; + } + } + if (guard.capability_id === "explore" && frontier.enabled === true && gap) { + if (gap.status === "scheduled" && gap.successor_todo_id) { + outcome = "new_runnable_successor"; capabilityOutcome = "new_runnable_composition_experiment"; + } else if (!repeated && typeof progressFingerprint === "string") { + result = array(gap.execution_results).find(row => row.progress_fingerprint === progressFingerprint + && row.todo_id === observation.work_item_id && researchDigest(row.progress) === researchDigest(observation)); + if (result && gap.status === "observed" && result.todo_id === gap.observed_todo_id + && progressFingerprint === gap.observed_progress_fingerprint) { + outcome = "capability_evidence_observed"; capabilityOutcome = "composition_experiment_observed"; + } else if (result && gap.status === "dismissed" && result.todo_id === gap.dismissed_todo_id + && result.disposition === "dismissed") { + outcome = "capability_evidence_observed"; capabilityOutcome = "composition_candidate_dismissed"; + } else if (result && gap.status === "deferred" && result.todo_id === gap.deferred_todo_id + && result.disposition === "deferred" && result.blocker_id === gap.blocker_todo_id + && !(Array.isArray(params.claimed_blocker_ids) && params.claimed_blocker_ids.includes(result.blocker_id))) { + outcome = "new_concrete_blocker"; capabilityOutcome = "composition_temporarily_deferred"; + } + } + } + return {schema_version: "replan_semantic_delta_v0", obligation_id: params.obligation_id, + accepted: outcome !== null, outcomes: outcome ? [outcome] : [], satisfying_outcomes: outcome ? [outcome] : [], + required_any_of: ["new_runnable_successor", "capability_evidence_observed", "new_concrete_blocker", "capability_duty_retired"], + capability_guard: guard, capability_outcome: capabilityOutcome, + ...(retirement ? {retirement_contract: retirement} : {}), + ...(retired ? {retirement: {...retirement, progress_fingerprint: progressFingerprint}} : {}), + ...(gap?.successor_todo_id ? {successor_todo_id: gap.successor_todo_id} : {}), + observation_fingerprint: result?.observation_fingerprint ?? null, + reason_code: outcome ? "research_semantic_delta_accepted" : !gap || frontier.enabled !== true + ? "research_frontier_invalidated" : repeated ? "research_result_replayed" : "research_result_required", + reason: outcome ? "the selected evidence duty has an exact canonical transition" + : "the selected evidence duty requires a current bound experiment result or runnable successor; reads, ACKs, stale inputs and unrelated progress cannot settle it"}; +} + +export function validateResearchCompositionSuccessor(params: JsonObject): JsonObject { + const frontier = requireJsonObject(params.frontier, "live research frontier"), todo = requireJsonObject(params.todo, "successor intent"); + const gap = object(frontier.selected_gap); + const refs = todo.explore_result_node_refs; + const accepted = frontier.enabled === true && gap.status === "pending" + && todo.replan_obligation_id === gap.obligation_id && boundTodo(todo, gap, gap.obligation_id, params.agent_id) + && ["open", "claimed"].includes(String(todo.status)) && !todo.resume_when + && Array.isArray(refs) && (gap.active_experiment_node_ids as string[]).includes(String(refs[0])); + if (!accepted) throw new EffectRuntimeRequestError( + `research successor ${gap.obligation_id ?? params.obligation_id}: bind one current binary experiment with joint_probe and no deferral; read Explore summary before todo add`); + return {accepted: true, gap_id: gap.gap_id, obligation_id: gap.obligation_id}; +} + +/** A completion guard is eligibility evidence, never a replacement for actor, + * lease or CAS admission. Historical terminal replays do not execute new work. */ +export function qualifyResearchCompletion(params: JsonObject): JsonObject { + const frontier = requireJsonObject(params.frontier, "research completion frontier"); + const todo = requireJsonObject(params.todo, "completion Todo"); + const gaps = array(frontier.lineage_gaps); + const required = frontier.enabled === true && todo.status !== "done" && todo.role !== "user" + && (todo.action_kind === "joint_probe" || todo.capability_binding_ref === "explore:research-composition-v0" + || gaps.some(gap => gap.obligation_id === todo.replan_obligation_id)); + if (!required) return {required: false, allowed: true, evidence: null}; + const actor = params.actor_agent_id ?? todo.claimed_by; + for (const gap of gaps) { + if (!["observed", "dismissed"].includes(String(gap.status)) || !boundTodo(todo, gap, gap.obligation_id, actor)) continue; + const result = array(gap.execution_results).find(row => row.todo_id === todo.todo_id + && ["exploration_exhausted", "no_followup"].includes(String(row.result_class)) + && (todo.explore_result_node_refs as string[]).includes(String(row.experiment_node_id))); + if (!result) continue; + return {required: true, allowed: true, evidence: { + schema_version: "capability_completion_evidence_v0", capability_id: "explore", goal_id: params.goal_id, + todo_id: todo.todo_id, agent_id: actor, gap_id: gap.gap_id, obligation_id: gap.obligation_id, + frontier_revision: gap.frontier_revision, experiment_node_id: result.experiment_node_id, + input_observations: gap.input_observations, observation_fingerprint: result.observation_fingerprint, + progress_fingerprint: result.progress_fingerprint, evidence_ids: result.evidence_ids, + disposition: result.disposition === "dismissed" ? "candidate_dismissed" : "experiment_observed", + }}; + } + return {required: true, allowed: false, reason_code: "research_experiment_result_required", + reason: "This Todo requires its current typed experiment observation with exact task/input lineage before closeout. Read Explore summary and record the result through explore observe.", + evidence: null}; +} + +/** The transport supplies a fresh canonical Todo snapshot and its existing + * actionable-open decision. Historical replay bypasses this gate without + * rewriting evidence; any later settlement must revalidate its own authority. + */ +export function validateResearchExecution(params: JsonObject): JsonObject { + const observation = validateResearchAttribution(params); + const lineage = requireJsonObject(observation.execution_lineage, "execution_lineage"); + const todo = requireJsonObject(params.todo, "canonical execution Todo"); + const node = (params.nodes as JsonObject[]).find(row => row.node_id === observation.explore_node_id); + const gap = researchCompositionGaps(params).find(row => row.gap_id === lineage.gap_id); + const reject = (reason: string): never => { + throw new EffectRuntimeRequestError(`research execution ${lineage.replan_obligation_id}: ${reason}; read the current Todo and Explore summary before explore observe`); + }; + if (lineage.goal_id !== params.goal_id || lineage.agent_id !== params.agent_id) { + reject("Goal or actor differs from execution lineage"); + } + if (todo.todo_id !== lineage.successor_todo_id || todo.claimed_by !== lineage.agent_id + || todo.replan_obligation_id !== lineage.replan_obligation_id + || todo.task_class !== "advancement_task" || todo.action_kind !== "joint_probe" + || (todo.role && todo.role !== "agent") + || todo.actionable_open !== true || todo.archive_state === "archive" + || (Array.isArray(todo.excluded_agents) && todo.excluded_agents.includes(lineage.agent_id))) { + reject("a current same-agent runnable joint-probe Todo with the exact obligation is required"); + } + const refs = todo.explore_result_node_refs; + if (!Array.isArray(refs) || refs.length !== 1 || refs[0] !== observation.explore_node_id + || (todo.target_key && todo.target_key !== observation.explore_node_id)) { + reject("Todo must bind exactly this experiment node"); + } + if (node?.node_kind !== "experiment" || (node.agent_id && node.agent_id !== lineage.agent_id) + || !gap || gap.state !== "pending" + || !(gap.experiment_node_ids as string[]).includes(String(observation.explore_node_id)) + || JSON.stringify(gap.input_observations) !== JSON.stringify(observation.input_observations)) { + reject("the experiment must cover this pending gap's current exact input observations"); + } + const progress = observation.progress as JsonObject; + if (progress.result_class === "unchanged" || !(progress.evidence_ids as string[]).length) { + reject("an execution result needs typed progress and evidence; a read or ACK is insufficient"); + } + return observation; +} diff --git a/loopx/control_plane/capabilities/explore_research_terminal.ts b/loopx/control_plane/capabilities/explore_research_terminal.ts new file mode 100644 index 0000000000..af08bab1bb --- /dev/null +++ b/loopx/control_plane/capabilities/explore_research_terminal.ts @@ -0,0 +1,93 @@ +/** Native completion's fixed graph IO adapter. The model supplies no process, + * snapshot, authority or approval fields; the provider supplies the actual Todo. */ +import {spawn, type ChildProcessWithoutNullStreams} from "node:child_process"; +import {readFile} from "node:fs/promises"; +import {createHash} from "node:crypto"; +import {isAbsolute} from "node:path"; +import {fileURLToPath} from "node:url"; +import {createInterface} from "node:readline"; +import type {JsonObject} from "../effect_program.ts"; +import type {TodoTerminalEvidenceGuard} from "../coordination/todo_terminal_lifecycle.ts"; +import {requireJsonObject} from "../runtime_decode.ts"; +import {normalizeResearchCompositionPolicy, qualifyResearchCompletion} from "./explore_research_execution.ts"; + +export class ResearchTerminalEvidenceHost { + private child: ChildProcessWithoutNullStreams | null = null; + private exited: Promise | null = null; + private readonly root: string; + private readonly goal: string; + private readonly source: JsonObject; + constructor(root: string, goal: string, source: JsonObject) {this.root = root; this.goal = goal; this.source = source;} + + readonly qualify: TodoTerminalEvidenceGuard = async context => { + if (context.todo.role === "user" || context.todo.status === "done" + || context.todo.action_kind !== "joint_probe" + && context.todo.capability_binding_ref !== "explore:research-composition-v0" + && !(Array.isArray(context.todo.explore_result_node_refs) && context.todo.explore_result_node_refs.length) + && !context.todo.replan_obligation_id) return {required: false, allowed: true, evidence: null}; + const source = requireJsonObject(this.source, "research registry source"); + const bytes = await readFile(String(source.path)); + if (createHash("sha256").update(bytes).digest("hex") !== source.sha256) { + return {allowed: false, reason_code: "authority_source_changed", reason: "Goal configuration changed before completion; read the current source."}; + } + const registry = requireJsonObject(JSON.parse(bytes.toString("utf8")), "research registry"); + const goals = Array.isArray(registry.goals) ? registry.goals : []; + const matches = goals.filter((value): value is JsonObject => value !== null && typeof value === "object" + && !Array.isArray(value)).filter(goal => goal.id === this.goal); + if (matches.length > 1) throw new Error("ambiguous research Goal source"); + const spawnPolicy = matches[0]?.spawn_policy as JsonObject | undefined; + const harness = spawnPolicy?.explore_harness as JsonObject | undefined; + if (!harness || !normalizeResearchCompositionPolicy({harness}).enabled) return {required: false, allowed: true, evidence: null}; + const python = process.env.LOOPX_EFFECT_RUNTIME_PYTHON; + if (!python || !isAbsolute(python) || python.includes("\0")) { + return {allowed: false, reason_code: "research_evidence_host_required", + reason: "Native research completion needs the source-selected Python adapter context; use the installed LoopX entrypoint."}; + } + this.child = spawn(python, ["-m", "loopx.capabilities.explore.research_snapshot_host"], { + stdio: ["pipe", "pipe", "pipe"], env: {...process.env, PYTHONPATH: fileURLToPath(new URL("../../../", import.meta.url))}, + }); + this.exited = new Promise(resolve => this.child!.once("close", () => resolve())); + this.child.on("error", () => {}); this.child.stdin.on("error", () => {}); this.child.stderr.resume(); + const replies = createInterface({input: this.child.stdout})[Symbol.asyncIterator](); + this.child.stdin.write(JSON.stringify({runtime_root: this.root, goal_id: this.goal, + harness, agent_id: context.actor_agent_id ?? context.todo.claimed_by, + todos: context.todos.map(todo => Object.fromEntries([ + "todo_id", "role", "status", "claimed_by", "replan_obligation_id", "task_class", "action_kind", + "archive_state", "excluded_agents", "explore_result_node_refs", "target_key", + "resume_when", "unblocks_todo_id", + ].filter(field => Object.hasOwn(todo, field)).map(field => [field, todo[field]]))), + }) + "\n"); + let timer: ReturnType | undefined; + try { + const line = await Promise.race([replies.next().then(row => row.done ? null : row.value), + this.exited.then(() => null), new Promise(resolve => {timer = setTimeout(() => resolve(null), 20000);})]); + if (line === null || Buffer.byteLength(line, "utf8") > 2 * 1024 * 1024) { + return {allowed: false, reason_code: "research_graph_snapshot_unavailable", reason: "Current locked research graph is unavailable; retry its source read before closeout."}; + } + const snapshot = requireJsonObject(JSON.parse(line), "research graph snapshot"); + if (snapshot.schema_version !== "research_graph_snapshot_v0" || snapshot.error) { + if (snapshot.error === "LockAcquireTimeoutError") return {allowed: false, reason_code: "research_graph_busy", + reason: "An Explore writer owns the graph lock; retry this same completion identity after the writer finishes."}; + return {allowed: false, reason_code: "research_graph_snapshot_invalid", reason: "The current research graph is invalid; repair its canonical source before closeout."}; + } + return qualifyResearchCompletion({frontier: snapshot.frontier, todo: context.todo, + actor_agent_id: context.actor_agent_id, goal_id: context.goal_id}); + } finally {clearTimeout(timer);} + }; + + current(): boolean { + // A child that exited no longer owns the kernel lock. Check this again at + // the transaction's existing source-admission boundaries, including CAS. + return this.child === null || this.child.exitCode === null && this.child.signalCode === null; + } + + async close(): Promise { + if (this.child) { + this.child.stdin.end("release\n"); + // The host exits after releasing its kernel lock. A bounded forced exit + // also closes the descriptors if its process cannot finish normally. + const timer = setTimeout(() => this.child?.kill(), 1000); + try {await this.exited;} finally {clearTimeout(timer);} + } + } +} diff --git a/loopx/control_plane/coordination/local_authority_runtime.ts b/loopx/control_plane/coordination/local_authority_runtime.ts index 60a53f4b15..c0fdb41fee 100644 --- a/loopx/control_plane/coordination/local_authority_runtime.ts +++ b/loopx/control_plane/coordination/local_authority_runtime.ts @@ -2,6 +2,7 @@ import {requirePromotionRegisteredAgents} from "./shadow_registry_source.ts"; import {readPromotionReceipt, commitPromotionAndReadBack} from './promotion_receipt.ts'; import {reviewedPromotionPlan, promotionPlanDigest, decodeReviewedPromotionOperation, REVIEWED_PROMOTION_OPERATION_RESULT_SCHEMA} from './reviewed_promotion_plan.ts'; import {registryAuthoritySourceCheck} from "./authority_source.ts"; +import {ResearchTerminalEvidenceHost} from "../capabilities/explore_research_terminal.ts"; import {decodeTaskLeaseProof} from "./task_lease_proof.ts"; import {COORDINATION_TODO_ARCHIVE_RESULT_SCHEMA} from "./todo_archive.ts"; import {readCoordinationOwnership} from "./ownership_observation.ts"; @@ -1290,6 +1291,8 @@ export async function terminalLifecycleLocalCoordinationTodo( const store = await openRuntimeStore(root, goalId, dependencies); sourceAuthority = sourceAuthorityFor(store); providerEvidence.source_authority = sourceAuthority; + const researchEvidence = new ResearchTerminalEvidenceHost(root, goalId, requireJsonObject(input.registry_source, "registry source")); + try { return {...await executeCoordinationTodoTerminalLifecycle(store, { validation_source_provider_revision: input.validation_source_provider_revision == null ? null : requireAuthorityStoreId(input.validation_source_provider_revision, "validation source provider revision"), @@ -1348,7 +1351,8 @@ export async function terminalLifecycleLocalCoordinationTodo( ? null : requireJsonObject(input.completion_policy_request, "completion_policy_request"), dry_run: input.dry_run as boolean, now: claimObservedAt(input.observed_at), - }, authoritySourcesCurrent), ...providerEvidence}; + }, async () => await authoritySourcesCurrent() && researchEvidence.current(), researchEvidence.qualify), ...providerEvidence}; + } finally {await researchEvidence.close();} }); } catch (error) { return {schema_version: COORDINATION_TODO_TERMINAL_LIFECYCLE_RESULT_SCHEMA, diff --git a/loopx/control_plane/coordination/todo_blocked_lifecycle.ts b/loopx/control_plane/coordination/todo_blocked_lifecycle.ts index d6610e0498..fb0f426583 100644 --- a/loopx/control_plane/coordination/todo_blocked_lifecycle.ts +++ b/loopx/control_plane/coordination/todo_blocked_lifecycle.ts @@ -1,4 +1,4 @@ -/** A user-directed pause is a Todo lifecycle change, not a completed delivery. +/** A bounded pause is a Todo lifecycle change, not a completed delivery. * Retire only an inactive execution generation in the same provider CAS; a * live holder must release its lease before another actor can pause the work. */ import type {JsonObject} from "../effect_program.ts"; @@ -7,17 +7,23 @@ import type {CoordinationTodoUpdateInput} from "./todo_update_intent.ts"; import {canonicalTaskLease} from "./task_lease_state.ts"; import {leaseEpoch, leaseIsActive, leaseVersion} from "../work_items/task_lease_acquire.ts"; import {releasedTaskLeaseRecord} from "../work_items/task_lease_lifecycle_decision.ts"; +import {normalizeTodoResumeWhen, TODO_RESUME_NORMALIZE_REQUEST_SCHEMA_VERSION} from "../todos/resume_condition.ts"; -const LIFECYCLE_FIELDS = new Set(["status", "reason", "clear_resume_when"]); +const LIFECYCLE_FIELDS = new Set(["status", "reason", "clear_resume_when", "resume_when"]); export function isBlockedLifecycleTransition( input: CoordinationTodoUpdateInput, todo: JsonObject, ): boolean { const intent = input.planning_intent ?? {}; + const pause = todo.status === "open" && intent.status === "blocked"; + const clearWait = intent.clear_resume_when === true && intent.resume_when == null; + const typedWait = pause && intent.clear_resume_when !== true && typeof intent.resume_when === "string" + && normalizeTodoResumeWhen({schema_version: TODO_RESUME_NORMALIZE_REQUEST_SCHEMA_VERSION, + resume_when: intent.resume_when}) === intent.resume_when; return todo.role === "agent" && - ((todo.status === "open" && intent.status === "blocked") || + (pause || (todo.status === "blocked" && intent.status === "open")) && - intent.clear_resume_when === true && + (clearWait || typedWait) && typeof intent.reason === "string" && Boolean(intent.reason.trim()) && Object.keys(input.patch).length === 0 && input.clear_fields.length === 0 && Object.keys(intent).every(field => LIFECYCLE_FIELDS.has(field)); diff --git a/loopx/control_plane/coordination/todo_terminal_lifecycle.ts b/loopx/control_plane/coordination/todo_terminal_lifecycle.ts index 70c3d4d05f..080566bbc7 100644 --- a/loopx/control_plane/coordination/todo_terminal_lifecycle.ts +++ b/loopx/control_plane/coordination/todo_terminal_lifecycle.ts @@ -1,5 +1,11 @@ import {planUserCompletion} from "../todos/user_completion.ts"; import {AUTHORITY_SOURCE_CHANGED, uncheckedAuthoritySource, type AuthoritySourceCheck} from "./authority_source.ts"; + +/** Service-owned evidence admission, evaluated against the actual provider + * head. Public requests cannot supply this callback or an approval boolean. */ +export type TodoTerminalEvidenceGuard = (context: { + goal_id: string; todo: JsonObject; todos: JsonObject[]; actor_agent_id: string | null; command: string; +}) => Promise; import {normalizeTodoUpdateInput, prepareUpdatedTodo, type CoordinationTodoUpdateInput, type TodoCompletionEdit} from "./todo_update_intent.ts"; import {todoUpdateAdmissionRejection} from "./todo_update_admission.ts"; import { createHash } from "node:crypto"; @@ -1023,6 +1029,7 @@ export async function executeCoordinationTodoTerminalLifecycle( store: AuthorityStore, rawInput: CoordinationTodoTerminalLifecycleInput, authoritySourcesCurrent: AuthoritySourceCheck = uncheckedAuthoritySource, + terminalEvidenceGuard: TodoTerminalEvidenceGuard | null = null, ): Promise { let normalized: CoordinationTodoTerminalLifecycleInput; try { @@ -1220,6 +1227,18 @@ export async function executeCoordinationTodoTerminalLifecycle( {goal_acceptance_guard: acceptance}, "decision_rejection"); } const acceptanceRequirements = acceptanceCompletionRequirements(completionHead, input.goal_id, input.todo_id); + let capabilityCompletionEvidence: JsonObject | null = null; + if (terminalEvidenceGuard !== null && !(authority.outcome === "no_change" && todo.status === "done")) { + const evidenceGuard = await terminalEvidenceGuard({goal_id: input.goal_id, todo, + todos: [...projection.todos.values()], actor_agent_id: input.actor_agent_id, command: input.command}); + if (evidenceGuard.allowed !== true) { + return terminalFailure(String(evidenceGuard.reason_code ?? "capability_completion_evidence_required"), + String(evidenceGuard.reason ?? "Current capability evidence is required before closeout"), + {capability_completion_guard: evidenceGuard}, "decision_rejection"); + } + capabilityCompletionEvidence = evidenceGuard.evidence == null ? null + : canonicalAuthorityObject(evidenceGuard.evidence, "capability completion evidence"); + } const acceptanceBinding = acceptanceRequirements === null ? null : acceptanceSourceBinding(input, acceptanceRequirements, head.provider_revision); let acceptanceEvidence: JsonObject | null = null; @@ -1590,6 +1609,7 @@ export async function executeCoordinationTodoTerminalLifecycle( completed_at: target.todo.completed_at, ...(completionResult === null ? {} : {completion_result: completionResult}), ...(acceptanceEvidence === null ? {} : {goal_acceptance_completion: acceptanceEvidence}), + ...(capabilityCompletionEvidence === null ? {} : {capability_completion_evidence: capabilityCompletionEvidence}), // A preview that omits this would show an unconditional close for work the // real call still gates. Name the criteria the real call must run; never // their argv, which stays out of every projection. diff --git a/loopx/control_plane/effect_runtime.py b/loopx/control_plane/effect_runtime.py index ed9ec18316..6acb872f96 100644 --- a/loopx/control_plane/effect_runtime.py +++ b/loopx/control_plane/effect_runtime.py @@ -8,6 +8,7 @@ import shutil import socket import subprocess +import sys import tempfile import time import uuid @@ -277,6 +278,10 @@ def _runtime_fingerprint_for_snapshot( snapshot: _RuntimeSourceSnapshot, ) -> str: digest = hashlib.sha256() + # The daemon's fixed Python adapter context must follow the interpreter + # selected by this caller, even when two environments share source files. + digest.update(sys.executable.encode("utf-8")) + digest.update(str(sys.version_info[:3]).encode("ascii")) source_root = Path(root) for relative, *_metadata in snapshot: digest.update(relative.encode("utf-8")) @@ -823,6 +828,8 @@ def _start_runtime(*, fingerprint: str, info_path: Path) -> dict[str, Any]: token = secrets.token_urlsafe(32) environment = os.environ.copy() environment["LOOPX_EFFECT_RUNTIME_TOKEN"] = token + # Fixed daemon context, never a model-supplied executable argument. + environment["LOOPX_EFFECT_RUNTIME_PYTHON"] = sys.executable # Capture stderr so a rejected startup can publish a typed # configuration diagnostic instead of a bare exit status. The capture # is an unlinked temporary file, so it cannot deadlock the child on a diff --git a/loopx/control_plane/effect_runtime_handlers.ts b/loopx/control_plane/effect_runtime_handlers.ts index c239fa3c49..5ace60d03b 100644 --- a/loopx/control_plane/effect_runtime_handlers.ts +++ b/loopx/control_plane/effect_runtime_handlers.ts @@ -1,6 +1,9 @@ import {manageNewGoalStorage} from "./coordination/local_authority_defaults.ts"; import {projectDecisionNotice} from "./presentation/decision_notice.ts"; import {normalizeResearchObservation, validateResearchAttribution, projectResearchFrontier} from "./capabilities/explore_research.ts"; +import {validateResearchExecution, normalizeResearchCompositionPolicy, researchCompositionFacts, + projectResearchComposition, qualifyResearchCompositionWriteback, + validateResearchCompositionSuccessor, qualifyResearchCompletion} from "./capabilities/explore_research_execution.ts"; import {projectTodoSummary} from "./todos/summary_projection.ts"; import {admitAutomationStart, confirmAutomationStart, manageAutomationCadence, projectCadenceSchedule} from "./quota/automation_cadence.ts"; import {deliverShadowEntry} from "./coordination/shadow_entry_delivery.ts"; @@ -863,6 +866,13 @@ export function createEffectRuntimeHandlers( ["work_item.replan_semantics.project", projectReplanSemantics], ["explore.research.normalize", normalizeResearchObservation], ["explore.research.validate_attribution", validateResearchAttribution], + ["explore.research.validate_execution", validateResearchExecution], + ["explore.research.composition_policy", normalizeResearchCompositionPolicy], + ["explore.research.composition_facts", researchCompositionFacts], + ["explore.research.composition", projectResearchComposition], + ["explore.research.composition_writeback", qualifyResearchCompositionWriteback], + ["explore.research.composition_successor", validateResearchCompositionSuccessor], + ["explore.research.completion", qualifyResearchCompletion], ["explore.research.frontier", projectResearchFrontier], ["work_item.replan_history.project", projectReplanHistory], ["work_item.replan_history.project_snapshot", projectReplanHistorySnapshot], diff --git a/loopx/control_plane/goals/goal_frontier/__init__.py b/loopx/control_plane/goals/goal_frontier/__init__.py index ea0b108b8e..ebfc350f53 100644 --- a/loopx/control_plane/goals/goal_frontier/__init__.py +++ b/loopx/control_plane/goals/goal_frontier/__init__.py @@ -1018,6 +1018,7 @@ def derive_goal_frontier_replan_obligation_from_summaries( current_transition_replan_ack: dict[str, Any] | None = None, acceptance_gaps: list[dict[str, Any]] | None = None, monitor_lane_semantically_valid: bool = True, + capability_obligation: dict[str, Any] | None = None, ) -> dict[str, Any] | None: """Return a compact replan obligation when the goal frontier has no advancement. @@ -1120,6 +1121,7 @@ def derive_goal_frontier_replan_obligation_from_summaries( current_agent_blocker_count=safe_non_negative_int( (agent_todo_summary or {}).get("current_agent_blocker_count") ), + capability_gap_pending=bool(capability_obligation and capability_obligation.get("required") is True), monitor_no_change_streak_triggered=( monitor_no_change_trigger is not None ), @@ -1139,6 +1141,8 @@ def derive_goal_frontier_replan_obligation_from_summaries( ) if not replan_rule.derives_obligation: return None + if replan_rule.rule is GoalFrontierReplanRule.CAPABILITY_EVIDENCE_GAP: + return capability_obligation if replan_rule.rule is GoalFrontierReplanRule.TODO_SUCCESSION_GAP: settlement_items = succession_gap_items[:3] settlement_todo_ids = [ @@ -1686,8 +1690,13 @@ def build_goal_frontier_projection_context_from_status( monitor_lane_semantically_valid=not goal_vision_state_is_closed( (latest_agent_vision or {}).get("state") ), + capability_obligation=(status_payload.get("bounded_research_frontier") or {}).get("obligation"), ) - frontier_transition_ack = replan_successor_transition_ack( + capability_transitions = (status_payload.get("bounded_research_frontier") or {}).get("settlement_transitions") or [] + frontier_transition_ack = next((candidate.get("ack") for candidate in capability_transitions + if candidate.get("obligation", {}).get("obligation_id") == (frontier_replan_obligation or {}).get("obligation_id")), None) if ( + frontier_replan_obligation or {} + ).get("capability_guard") else replan_successor_transition_ack( agent_todo_summary, agent_id=agent_id, replan_obligation=frontier_replan_obligation, @@ -1773,7 +1782,7 @@ def build_goal_frontier_projection_context_from_status( "projected_replan_ack": projected_replan_ack, "replan_transition_ack": replan_transition_ack, "run_replan_transition_ack": run_replan_transition_ack, - "replan_transition_candidates": [run_transition_candidate, frontier_transition_candidate], + "replan_transition_candidates": [run_transition_candidate, frontier_transition_candidate, *capability_transitions], } diff --git a/loopx/control_plane/goals/goal_frontier/replan_rules.py b/loopx/control_plane/goals/goal_frontier/replan_rules.py index 0b105b10c2..9ab75c8f1d 100644 --- a/loopx/control_plane/goals/goal_frontier/replan_rules.py +++ b/loopx/control_plane/goals/goal_frontier/replan_rules.py @@ -28,8 +28,11 @@ class GoalFrontierReplanRule(str, Enum): DUE_MONITOR_EXECUTION = "due_monitor_execution" FUTURE_MONITOR_WAIT = "future_monitor_wait" MONITOR_FRONTIER_EXHAUSTED = "monitor_frontier_exhausted" + CAPABILITY_EVIDENCE_GAP = "capability_evidence_gap" +# Stable presentation indices for the v0 wire contract. Optional rules append +# here; their actual precedence is the explicit table in the interpreter. GOAL_FRONTIER_REPLAN_RULE_ORDER = tuple(GoalFrontierReplanRule) @@ -51,6 +54,7 @@ class GoalFrontierReplanFacts: outcome_checkpoint_replan_required: bool = False long_todo_chain_triggered: bool = False current_agent_blocker_count: int = 0 + capability_gap_pending: bool = False monitor_no_change_streak_triggered: bool = False monitor_only_lane: bool = False monitor_count: int = 0 @@ -143,6 +147,12 @@ def select_goal_frontier_replan_rule( False, "an explicit current-agent blocker owns the empty frontier", ), + ( + GoalFrontierReplanRule.CAPABILITY_EVIDENCE_GAP, + facts.capability_gap_pending and facts.selectable_frontier_advancement == 0, + True, + "a caller-owned evidence gap remains without selectable advancement", + ), ( GoalFrontierReplanRule.MONITOR_NO_CHANGE_STREAK, facts.monitor_only_lane diff --git a/loopx/control_plane/quota/heartbeat_receipt.py b/loopx/control_plane/quota/heartbeat_receipt.py index f02fec2e14..e350d203c5 100644 --- a/loopx/control_plane/quota/heartbeat_receipt.py +++ b/loopx/control_plane/quota/heartbeat_receipt.py @@ -172,6 +172,7 @@ def ensure_turn_heartbeat_settlement_receipt( *, semantic_replan_guard_scoped: bool, semantic_replan_obligation_id: str | None, + semantic_replan_capability_guard: Mapping[str, object] | None = None, ) -> dict[str, object]: """Idempotently bind a Turn-created quota guard to its settlement identity. @@ -239,6 +240,14 @@ def ensure_turn_heartbeat_settlement_receipt( raise HeartbeatReceiptIdentityConflictError( "Turn heartbeat receipt belongs to another semantic replan guard" ) + if semantic_replan_capability_guard is not None and any(effective_details.get( + source + ) != semantic_replan_capability_guard[field] for source, field in ( + ("semantic_replan_capability_id", "capability_id"), + ("semantic_replan_gap_id", "gap_id"), + ("semantic_replan_frontier_revision", "frontier_revision"), + )): + raise HeartbeatReceiptIdentityConflictError("Turn heartbeat receipt belongs to another capability guard") return effective details = { @@ -253,6 +262,10 @@ def ensure_turn_heartbeat_settlement_receipt( details["semantic_replan_obligation_id"] = ( normalized_semantic_replan_obligation_id or "" ) + if semantic_replan_capability_guard is not None: + details["semantic_replan_capability_id"] = semantic_replan_capability_guard["capability_id"] + details["semantic_replan_gap_id"] = semantic_replan_capability_guard["gap_id"] + details["semantic_replan_frontier_revision"] = semantic_replan_capability_guard["frontier_revision"] source_event_id = ( str(effective.get("event_id") or "").strip() if effective is not None @@ -534,6 +547,13 @@ def heartbeat_receipt_view( receipt["semantic_replan_obligation_id"] = ( semantic_replan_obligation_id ) + if details.get("semantic_replan_capability_id"): + receipt["semantic_replan_capability_guard"] = { + "schema_version": "semantic_replan_capability_guard_v0", + "capability_id": details["semantic_replan_capability_id"], + "gap_id": details.get("semantic_replan_gap_id"), + "frontier_revision": details.get("semantic_replan_frontier_revision"), + } pending_action_todo_id = heartbeat_receipt_pending_action_todo_id(event) if pending_action_todo_id: receipt["pending_action_selection"] = { diff --git a/loopx/control_plane/quota/settlement.py b/loopx/control_plane/quota/settlement.py index f5317506cc..ab8757983c 100644 --- a/loopx/control_plane/quota/settlement.py +++ b/loopx/control_plane/quota/settlement.py @@ -151,7 +151,7 @@ class QuotaSettlementReadback: terminal_closeout: SettlementResult[dict[str, Any]] terminal_settlement: SettlementResult[dict[str, Any]] workspace_causality: dict[str, str] | None - semantic_replan_guard: dict[str, str | None] | None + semantic_replan_guard: dict[str, Any] | None writeback_run: dict[str, Any] | None spend_run: dict[str, Any] | None heartbeat_receipt: dict[str, Any] | None @@ -217,6 +217,8 @@ def render_settlement_progress_markdown(payload: dict[str, Any]) -> list[str]: lines = [f"- settlement: `{progress.get('state')}`"] if progress.get("closeout_kind") == "typed_blocked_writeback_no_spend": lines.append("- closeout: typed blocked writeback; no quota slot spent") + elif progress.get("closeout_kind") == "capability_duty_retired_no_spend": + lines.append("- closeout: exact invalidated capability duty retired; no quota slot spent") owed = payload.get("settlement_owed") if isinstance(owed, dict): lines.extend([f"- settlement_owed: {owed['reason']}", "", "```sh", owed["command"], "```"]) @@ -272,7 +274,7 @@ def _optional_readback_record(value: Any) -> dict[str, Any] | None: return dict(value) -def _semantic_replan_guard(value: Any) -> dict[str, str | None] | None: +def _semantic_replan_guard(value: Any) -> dict[str, Any] | None: guard = _optional_readback_record(value) if guard is None: return None @@ -292,11 +294,19 @@ def _semantic_replan_guard(value: Any) -> dict[str, str | None] | None: or legacy_guard_claims_selection ): raise RuntimeError("TypeScript semantic replan guard shape mismatch") - return { + result: dict[str, Any] = { "schema_version": SEMANTIC_REPLAN_GUARD_SCHEMA, "scope": str(scope), "selected_obligation_id": selected_obligation_id, } + if "selected_capability_guard" in guard: + capability = guard["selected_capability_guard"] + if not isinstance(capability, dict) or selected_obligation_id is None: + raise RuntimeError("selected capability guard requires a typed object and obligation") + # The TS readback owner validates the schema and ids. Preserve the + # selected authority evidence when adapting the receipt to refresh. + result["selected_capability_guard"] = dict(capability) + return result def read_heartbeat_settlement( diff --git a/loopx/control_plane/quota/settlement_cli.py b/loopx/control_plane/quota/settlement_cli.py index 2096d20ca4..a2c8b1cef0 100644 --- a/loopx/control_plane/quota/settlement_cli.py +++ b/loopx/control_plane/quota/settlement_cli.py @@ -375,6 +375,15 @@ def quota_rollout_details( "quiet_noop_allowed": bool(agent_channel.get("quiet_noop_allowed")), "closeout_required": closeout_required, } + replan_packet = payload.get("replan_action_packet") + receipt = payload.get("heartbeat_receipt") + capability_guard = (replan_packet.get("capability_guard") if isinstance(replan_packet, Mapping) else None) or ( + receipt.get("semantic_replan_capability_guard") if isinstance(receipt, Mapping) else None + ) + if isinstance(capability_guard, Mapping): + details["semantic_replan_capability_id"] = capability_guard["capability_id"] + details["semantic_replan_gap_id"] = capability_guard["gap_id"] + details["semantic_replan_frontier_revision"] = capability_guard["frontier_revision"] cli_channel = interaction.get("cli_channel") if isinstance(cli_channel, Mapping) and cli_channel.get("quota_spend_source"): details["quota_spend_source"] = cli_channel["quota_spend_source"] diff --git a/loopx/control_plane/quota/settlement_phase.ts b/loopx/control_plane/quota/settlement_phase.ts index b2ff8d2efd..9c7e75a2e2 100644 --- a/loopx/control_plane/quota/settlement_phase.ts +++ b/loopx/control_plane/quota/settlement_phase.ts @@ -17,6 +17,45 @@ export function isBoundedBlockedRetry(value: unknown, todoId: string | null): bo return Number.isFinite(delay) && delay >= 60 && delay <= 30 * 60; } +/** A service-qualified capability duty can retire its exact admitted Turn. + * The caller first verifies the durable writeback receipt. No Task/Goal + * completion, progress or debit follows from this lifecycle evidence. */ +export function isCapabilityRetirementWriteback(value: unknown, identity: SettlementIdentity, selectedGuard: unknown): boolean { + const run = jsonObject(value), guard = jsonObject(selectedGuard); + const ack = jsonObject(run?.autonomous_replan_ack), delta = jsonObject(ack?.semantic_delta); + const retirement = jsonObject(delta?.retirement), original = jsonObject(retirement?.original_guard); + const deltaGuard = jsonObject(delta?.capability_guard), progress = jsonObject(run?.progress_observation); + const sameGuard = (candidate: Record | null) => !!guard && !!candidate + && candidate.schema_version === "semantic_replan_capability_guard_v0" + && candidate.capability_id === guard.capability_id && candidate.gap_id === guard.gap_id + && candidate.frontier_revision === guard.frontier_revision; + return identity.binding_kind === "autonomous_replan" && identity.replan_obligation_id !== null + && !!run && run.goal_id === identity.goal_id && run.agent_id === identity.agent_id + && run.turn_instance_id === identity.turn_instance_id && run.replan_obligation_id === identity.replan_obligation_id + && run.delivery_outcome === "outcome_gap" && ack?.recorded === true + && delta?.schema_version === "replan_semantic_delta_v0" && delta.accepted === true + && delta.obligation_id === identity.replan_obligation_id + && JSON.stringify(delta.outcomes) === '["capability_duty_retired"]' + && JSON.stringify(delta.satisfying_outcomes) === '["capability_duty_retired"]' + && retirement?.schema_version === "capability_obligation_retirement_v0" && retirement.disposition === "invalidated" + && retirement.obligation_id === identity.replan_obligation_id && retirement.capability_id === guard?.capability_id + && ["source_disabled", "source_ineligible", "source_revision_changed"].includes(String(retirement.reason_code)) + && retirement.blocking_todo_count === 0 && Array.isArray(retirement.blocking_todo_ids) && retirement.blocking_todo_ids.length === 0 + && typeof retirement.current_revision === "string" && retirement.current_revision.length > 0 + && sameGuard(original) && sameGuard(deltaGuard) + && progress?.schema_version === "typed_progress_observation_v0" && progress.result_class === "blocked" + && progress.work_item_id === identity.replan_obligation_id + && typeof progress.blocker_id === "string" && progress.blocker_id.length > 0 + && typeof progress.fingerprint === "string" && progress.fingerprint.length > 0 + && retirement.progress_fingerprint === progress.fingerprint + && jsonObject(retirement.progress_observation)?.blocker_id === progress.blocker_id + && jsonObject(retirement.progress_observation)?.work_item_id === identity.replan_obligation_id + && jsonObject(retirement.progress_observation)?.result_class === "blocked" + && jsonObject(retirement.progress_observation)?.schema_version === "typed_progress_observation_v0" + && JSON.stringify(jsonObject(retirement.progress_observation)?.evidence_ids) === JSON.stringify(progress.evidence_ids) + && JSON.stringify(progress.evidence_ids) === JSON.stringify([retirement.current_revision]); +} + /** The committed checkpoint accepts progress for a Turn, not Todo completion. * The caller must first verify this writeback's exact durable receipt. */ export function isAcceptedInFlightWriteback( diff --git a/loopx/control_plane/quota/settlement_readback.ts b/loopx/control_plane/quota/settlement_readback.ts index 8c2dbaab22..aadad7dfcf 100644 --- a/loopx/control_plane/quota/settlement_readback.ts +++ b/loopx/control_plane/quota/settlement_readback.ts @@ -33,6 +33,7 @@ import { isBoundedBlockedRetry, isCommittedMonitorPollEffect, isAcceptedInFlightWriteback, + isCapabilityRetirementWriteback, receiptBoundMonitorPhase, receiptBoundReplayPhase, } from "./settlement_phase.ts"; @@ -118,7 +119,7 @@ function settlementProgress( identity: SettlementResult, writeback: SettlementResult, spend: SettlementResult, writebackRun: JsonObject | null, spendRun: JsonObject | null, spendSource: unknown = "heartbeat", - blockedNoSpend = false, + noSpendKind: "typed_blocked_writeback_no_spend" | "capability_duty_retired_no_spend" | null = null, ): JsonObject { const source = spendSource ?? "heartbeat"; if (source !== "heartbeat" && source !== "visible-goal") { @@ -126,16 +127,16 @@ function settlementProgress( } const state: SettlementProgressState = identity.failure ? "identity_required" : writeback.failure ? (writebackRun ? "writeback_receipt_required" : "writeback_required") - : blockedNoSpend ? "settled" + : noSpendKind ? "settled" : spend.failure ? (spendRun ? "spend_receipt_required" : "spend_required") : "settled"; return { schema_version: "quota_settlement_progress_v0", state, next_step: identity.failure ? "validation" : writeback.failure ? "durable_writeback" - : blockedNoSpend ? null : spend.failure ? "quota_spend" : null, + : noSpendKind ? null : spend.failure ? "quota_spend" : null, quota_spend_source: source, - ...(blockedNoSpend ? { - closeout_kind: "typed_blocked_writeback_no_spend", + ...(noSpendKind ? { + closeout_kind: noSpendKind, } : {}), }; } @@ -406,7 +407,10 @@ function runEffectMatches( export function projectSemanticReplanGuard( receiptDetails: JsonObject, ): JsonObject { + const hasCapability = ["semantic_replan_capability_id", "semantic_replan_gap_id", "semantic_replan_frontier_revision"] + .some(field => Object.hasOwn(receiptDetails, field)); if (!Object.hasOwn(receiptDetails, "semantic_replan_obligation_id")) { + if (hasCapability) throw new EffectRuntimeRequestError("a capability guard requires a selected obligation", "malformed_settlement_state"); return { schema_version: SEMANTIC_REPLAN_GUARD_SCHEMA, scope: "legacy_unscoped", @@ -415,6 +419,9 @@ export function projectSemanticReplanGuard( } const rawObligationId = receiptDetails.semantic_replan_obligation_id; if (rawObligationId === "" || rawObligationId === null) { + if (hasCapability) { + throw new EffectRuntimeRequestError("a capability guard requires a selected obligation", "malformed_settlement_state"); + } return { schema_version: SEMANTIC_REPLAN_GUARD_SCHEMA, scope: "turn_guard", @@ -432,9 +439,26 @@ export function projectSemanticReplanGuard( schema_version: SEMANTIC_REPLAN_GUARD_SCHEMA, scope: "turn_guard", selected_obligation_id: selectedObligationId, + ...(hasCapability ? { + selected_capability_guard: decodeCapabilityGuard({schema_version: "semantic_replan_capability_guard_v0", + capability_id: receiptDetails.semantic_replan_capability_id, gap_id: receiptDetails.semantic_replan_gap_id, + frontier_revision: receiptDetails.semantic_replan_frontier_revision}), + } : {}), }; } +function decodeCapabilityGuard(value: unknown): JsonObject { + const guard = jsonObject(value); + if (!guard || guard.schema_version !== "semantic_replan_capability_guard_v0" + || typeof guard.capability_id !== "string" || !/^[a-z][a-z0-9-]{0,63}$/.test(guard.capability_id) + || typeof guard.gap_id !== "string" || !/^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/.test(guard.gap_id) + || typeof guard.frontier_revision !== "string" || !/^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/.test(guard.frontier_revision) + || Object.keys(guard).some(key => !["schema_version", "capability_id", "gap_id", "frontier_revision"].includes(key))) { + throw new EffectRuntimeRequestError("heartbeat receipt capability guard is malformed", "malformed_settlement_state"); + } + return guard; +} + function runMatchesBinding(run: JsonObject, identity: SettlementIdentity): boolean { return normalizeTodoId(run.todo_id) === identity.todo_id && normalizeReplanObligationId(run.replan_obligation_id) === @@ -1008,6 +1032,9 @@ function readQuotaSettlementFromRequest( const writeback = writebackResult(identity, writebackRun, writebackEvent); const spend = spendResult(identity, spendRun, spendEvent); + const semanticReplanGuard = projectSemanticReplanGuard(receiptDetails); + const retiredNoSpend = writeback.failure === null && spendRun === null && spendEvent === null + && isCapabilityRetirementWriteback(writebackRun, identity, semanticReplanGuard.selected_capability_guard); // The exact Turn-bound blocked writeback is itself a durable no-spend // closeout. It cannot certify Todo completion or become delivery progress. // A spend already committed for this identity remains an ordinary spend @@ -1024,7 +1051,8 @@ function readQuotaSettlementFromRequest( ); const terminalCloseout = terminalResult(identity, completionEvent); const withWriteback = settlementBindReduce(identityResult, writeback); - const settled = blockedNoSpend ? withWriteback : settlementBindReduce(withWriteback, spend); + const noSpendKind = retiredNoSpend ? "capability_duty_retired_no_spend" : blockedNoSpend ? "typed_blocked_writeback_no_spend" : null; + const settled = noSpendKind ? withWriteback : settlementBindReduce(withWriteback, spend); const terminalSettlement = settlementBindReduce(settled, terminalCloseout); const monitorPoll = committedMonitorPollFromSnapshot(snapshot, identity); const nestedCausality = typeof receiptDetails.delivery_workspace_causality === "object" && @@ -1042,7 +1070,6 @@ function readQuotaSettlementFromRequest( const workspaceCausality: DeliveryWorkspaceCausality | null = normalizeDeliveryWorkspaceCausality(nestedCausality, identity.todo_id) ?? normalizeDeliveryWorkspaceCausality(flatCausality, identity.todo_id); - const semanticReplanGuard = projectSemanticReplanGuard(receiptDetails); const todoBoundReplan = identity.binding_kind === "todo" && semanticReplanGuard.scope === "turn_guard" && semanticReplanGuard.selected_obligation_id !== null; @@ -1056,7 +1083,7 @@ function readQuotaSettlementFromRequest( optionalString(supersedeEvent.todo_id) === identity.todo_id, durable_writeback_present: writeback.failure === null, quota_spend_present: spend.failure === null, - no_spend_closeout_present: blockedNoSpend, + no_spend_closeout_present: noSpendKind !== null, }); const recovery = request.refresh_retry === null ? null : refreshRecovery( @@ -1081,7 +1108,7 @@ function readQuotaSettlementFromRequest( terminal_closeout: bundle(terminalCloseout), terminal_settlement: bundle(terminalSettlement), progress: settlementProgress(identityResult, writeback, spend, writebackRun, spendRun, - receiptDetails.quota_spend_source ?? spendRun?.source, blockedNoSpend), + receiptDetails.quota_spend_source ?? spendRun?.source, noSpendKind), workspace_causality: workspaceCausality, semantic_replan_guard: semanticReplanGuard, writeback_run: writebackRun, diff --git a/loopx/control_plane/quota/should_run_packet.py b/loopx/control_plane/quota/should_run_packet.py index 41ede57316..2a0774c936 100644 --- a/loopx/control_plane/quota/should_run_packet.py +++ b/loopx/control_plane/quota/should_run_packet.py @@ -1434,9 +1434,8 @@ def _build_active_quota_payload( next_action_warning=route.next_action_warning, replan_obligation=prepared.replan_obligation, ) - bounded_research_frontier = _dict_field( - prepared.status_payload, "bounded_research_frontier" - ) + frontier = _dict_field(prepared.status_payload, "bounded_research_frontier") + bounded_research_frontier = _dict_field(frontier or {}, "public_projection") or frontier _attach_truthy_fields( payload, bounded_research_frontier=bounded_research_frontier, @@ -1557,7 +1556,8 @@ def _build_settled_quota_payload( missing_gates=prepared.item.get("missing_gates"), agent_todo_summary=compact_quota_todo_summary_for_payload(prepared.agent_todo_summary) if prepared.agent_todo_summary else None, user_todo_summary=compact_quota_todo_summary_for_payload(prepared.user_todo_summary) if prepared.user_todo_summary else None, - bounded_research_frontier=_dict_field(prepared.status_payload, "bounded_research_frontier"), + bounded_research_frontier=(_dict_field(_dict_field(prepared.status_payload, "bounded_research_frontier") or {}, "public_projection") + or _dict_field(prepared.status_payload, "bounded_research_frontier")), ) if prepared.agent_scoped_user_todo_override: payload[str(prepared.agent_scoped_user_todo_override["kind"])] = prepared.agent_scoped_user_todo_override diff --git a/loopx/control_plane/quota/slot_accounting.py b/loopx/control_plane/quota/slot_accounting.py index 387ed81304..1f5e1fc77e 100644 --- a/loopx/control_plane/quota/slot_accounting.py +++ b/loopx/control_plane/quota/slot_accounting.py @@ -182,13 +182,15 @@ def _resolve_preview_settlement( if ( isinstance(readback.progress, dict) and readback.progress.get("closeout_kind") - == "typed_blocked_writeback_no_spend" + in {"typed_blocked_writeback_no_spend", "capability_duty_retired_no_spend"} ): return { "identity": readback.identity.value, "result": readback.settlement, "delivery_run": readback.writeback_run, "reason": ( + "this exact capability duty was retired and must not consume a quota slot; reassess the current frontier in a new Turn" + if readback.progress.get("closeout_kind") == "capability_duty_retired_no_spend" else "this Turn already closed with an exact typed blocked writeback " "and must not consume a quota slot; retry the Todo only after " "its external blocker changes or a bounded backoff" diff --git a/loopx/control_plane/quota/unsettled_host_turn_recovery.ts b/loopx/control_plane/quota/unsettled_host_turn_recovery.ts index c892b37277..0601c52091 100644 --- a/loopx/control_plane/quota/unsettled_host_turn_recovery.ts +++ b/loopx/control_plane/quota/unsettled_host_turn_recovery.ts @@ -63,6 +63,7 @@ const MISSING_RECEIPT_NAMES = [WRITEBACK_RECEIPT, SPEND_RECEIPT] as const; export const ACCEPTED_CLOSEOUTS = [ "validated_writeback_and_quota_spend", "typed_blocked_writeback_no_spend", + "capability_duty_retired_no_spend", "exact_committed_quota_monitor_poll", "typed_external_wait_with_runnable_successor", "typed_blocker_or_lifecycle_transition", @@ -255,8 +256,8 @@ export async function preflightPriorHostTurnCloseout( newestSettledTurn ??= selected.prior_turn_instance_id; if (newestSettledTurn === selected.prior_turn_instance_id) { const progress = jsonObject(readback.progress); - if (progress?.closeout_kind === "typed_blocked_writeback_no_spend") { - newestAcceptedCloseout = "typed_blocked_writeback_no_spend"; + if (progress?.closeout_kind === "typed_blocked_writeback_no_spend" || progress?.closeout_kind === "capability_duty_retired_no_spend") { + newestAcceptedCloseout = progress.closeout_kind; } } continue; diff --git a/loopx/control_plane/status/agent_lane_projection.py b/loopx/control_plane/status/agent_lane_projection.py index 88cf37c9f7..58d7fb965c 100644 --- a/loopx/control_plane/status/agent_lane_projection.py +++ b/loopx/control_plane/status/agent_lane_projection.py @@ -30,6 +30,7 @@ "agent_reward_memory", "autonomous_replan_ack", "autonomous_replan_obligation", + "bounded_research_frontier", "completed_todo_archive_warning", "control_plane", "external_progress_review", diff --git a/loopx/control_plane/work_items/progress_observation.py b/loopx/control_plane/work_items/progress_observation.py index 618b2dd0e3..871ebcd7d5 100644 --- a/loopx/control_plane/work_items/progress_observation.py +++ b/loopx/control_plane/work_items/progress_observation.py @@ -645,6 +645,9 @@ def build_replan_action_packet( ProgressResultClass.NO_FOLLOWUP.value, ], } + if isinstance(obligation.get("capability_guard"), Mapping): + packet["capability_guard"] = dict(obligation["capability_guard"]) + packet["allowed_terminal"] = [] if isinstance(selected_gap, Mapping): packet["bounded_frontier"] = { key: selected_gap[key] @@ -654,6 +657,8 @@ def build_replan_action_packet( "experiment_node_ref", "input_node_refs", "required_outcome", + "input_observations", + "frontier_revision", ) if key in selected_gap } diff --git a/loopx/control_plane/work_items/replan_semantics.ts b/loopx/control_plane/work_items/replan_semantics.ts index 36371e6df3..7062eab8cc 100644 --- a/loopx/control_plane/work_items/replan_semantics.ts +++ b/loopx/control_plane/work_items/replan_semantics.ts @@ -11,8 +11,10 @@ const VISION_OUTCOMES = [ "fresh_vision_path_outcome", "new_runnable_successor", "new_concrete_blocker", "coverage_backed_exploration_exhausted", "coverage_backed_no_followup", ] as const; -type SemanticOutcome = typeof PROGRESS_OUTCOMES[number] | typeof VISION_OUTCOMES[number]; -const KNOWN_OUTCOMES: ReadonlySet = new Set([...PROGRESS_OUTCOMES, ...VISION_OUTCOMES]); +// Capability-owned evidence is never a default generic progress exit. +const CAPABILITY_OUTCOMES = ["capability_evidence_observed", "capability_duty_retired"] as const; +type SemanticOutcome = typeof PROGRESS_OUTCOMES[number] | typeof VISION_OUTCOMES[number] | typeof CAPABILITY_OUTCOMES[number]; +const KNOWN_OUTCOMES: ReadonlySet = new Set([...PROGRESS_OUTCOMES, ...VISION_OUTCOMES, ...CAPABILITY_OUTCOMES]); // Outcomes that a renamed identifier alone can produce. An external progress // review found the evaluated identifiers not serving the goal, so for that // source they discharge only behind evidence ids absent from the whole @@ -151,6 +153,10 @@ export function requiredSemanticOutcomes(obligation: JsonObject): SemanticOutcom if (declared.some(value => !KNOWN_OUTCOMES.has(value))) { throw new EffectRuntimeRequestError("satisfying_semantic_outcomes contains an unknown typed outcome"); } + if (declared.some(value => (CAPABILITY_OUTCOMES as readonly string[]).includes(value)) && + object(obligation.capability_guard).schema_version !== "semantic_replan_capability_guard_v0") { + throw new EffectRuntimeRequestError("capability evidence requires a bound owning capability"); + } if (acceptanceHold && declared.some(value => !["new_runnable_successor", "new_concrete_blocker"].includes(value))) { throw new EffectRuntimeRequestError("acceptance recovery cannot widen its typed outcomes"); } @@ -220,6 +226,9 @@ export function projectReplanSemantics(value: unknown): JsonObject { const patch = object(vision.vision_patch); const path = object(vision.path_delta); let outcomes = strings(observation.delta_kinds); + if (outcomes.some(value => (CAPABILITY_OUTCOMES as readonly string[]).includes(value))) { + throw new EffectRuntimeRequestError("capability evidence must be qualified by its owning capability, not generic progress"); + } if (outcomes.some(outcome => !KNOWN_OUTCOMES.has(outcome))) { throw new EffectRuntimeRequestError("observation_delta contains an unknown typed outcome"); } diff --git a/loopx/control_plane/work_items/semantic_replan_writeback.py b/loopx/control_plane/work_items/semantic_replan_writeback.py index 0f7047df9c..85051c8e14 100644 --- a/loopx/control_plane/work_items/semantic_replan_writeback.py +++ b/loopx/control_plane/work_items/semantic_replan_writeback.py @@ -3,7 +3,7 @@ from __future__ import annotations import shlex -from collections.abc import Mapping +from collections.abc import Callable, Mapping from dataclasses import dataclass from typing import Any @@ -42,6 +42,14 @@ REPLAN_WRITEBACK_REJECTION_SCHEMA_VERSION = "replan_writeback_rejection_v0" +@dataclass(frozen=True) +class CapabilityReplanEvidence: + """Composition-root supplied facts and their capability-owned qualifier.""" + frontier: dict[str, Any] + qualify: Callable[[Mapping[str, Any], str, dict[str, Any] | None, list[dict[str, Any]]], + tuple[dict[str, Any], dict[str, Any]]] + + @dataclass(frozen=True) class RefreshReplanQualification: repair_delta_contract: dict[str, Any] | None @@ -119,19 +127,26 @@ def project_replan_writeback_rejection( runtime_root=runtime_root, ) ) + capability_guard = obligation.get("capability_guard") + if isinstance(capability_guard, Mapping) and rejection.semantic_delta.get("readback_actions"): + next_cli_actions = rejection.semantic_delta["readback_actions"] return { "schema_version": REPLAN_WRITEBACK_REJECTION_SCHEMA_VERSION, "required": True, "host_action": ( "settle_todo_lifecycle" if lifecycle_reentry is not None + else "read_current_capability_evidence" if capability_guard else "write_typed_semantic_delta" ), "obligation_id": obligation.get("obligation_id"), "resolution_mode": obligation.get("resolution_mode"), + **({"retirement_contract": rejection.semantic_delta["retirement_contract"]} + if rejection.semantic_delta.get("retirement_contract") else {}), "reason_code": rejection.semantic_delta.get("reason_code"), "triggers": triggers, "next_cli_actions": next_cli_actions, + **({"capability_guard": dict(capability_guard)} if isinstance(capability_guard, Mapping) else {}), } @@ -200,6 +215,8 @@ def qualify_replan_writeback( external_progress_review: Mapping[str, Any] | None = None, guard_scoped: bool = False, guard_semantic_replan_obligation_id: str | None = None, + capability_evidence: CapabilityReplanEvidence | None = None, + guard_capability: Mapping[str, Any] | None = None, ) -> tuple[dict[str, Any] | None, dict[str, Any] | None]: """Return the shared open obligation and the writeback's typed delta. @@ -261,12 +278,15 @@ def qualify_replan_writeback( "run_history": { "goals": [ { + **(registry_goal or {}), "id": goal_id, "latest_runs": list(newest_first_runs or []), } ] } } + if capability_evidence is not None: + status_payload["bounded_research_frontier"] = capability_evidence.frontier context = build_goal_frontier_projection_context_from_status( goal_id=goal_id, agent_id=safe_agent_id, @@ -297,6 +317,14 @@ def qualify_replan_writeback( else None ), ) + obligation = context.get("replan_obligation") + selected_capability = guard_capability if guard_scoped else (obligation or {}).get("capability_guard") + if selected_capability is not None: + if capability_evidence is None: + raise ValueError("selected capability writeback owner is not available") + selected_id = guard_semantic_replan_obligation_id if guard_scoped else (obligation or {}).get("obligation_id") + return capability_evidence.qualify(selected_capability, str(selected_id), progress_observation, + [run for run in newest_first_runs or [] if run.get("agent_id") == safe_agent_id]) if guard_scoped and guard_semantic_replan_obligation_id: transition_delta = guarded_replan_transition_delta( guard_scoped=guard_scoped, @@ -364,6 +392,8 @@ def enforce_open_replan_writeback( guard_semantic_replan_obligation_id: str | None = None, todo_fields: dict[str, Any] | None = None, external_progress_review: Mapping[str, Any] | None = None, + capability_evidence: CapabilityReplanEvidence | None = None, + guard_capability: Mapping[str, Any] | None = None, ) -> dict[str, Any] | None: """Fail closed unless concrete typed evidence satisfies the selected replan. @@ -388,6 +418,7 @@ def enforce_open_replan_writeback( todo_fields=todo_fields, guard_scoped=guard_scoped, guard_semantic_replan_obligation_id=guard_semantic_replan_obligation_id, + capability_evidence=capability_evidence, guard_capability=guard_capability, ) if not obligation: if isinstance(semantic_delta, dict) and semantic_delta.get("accepted") is True: @@ -444,6 +475,8 @@ def qualify_refresh_replan_writeback( classification: str, delivery_outcome: str | None, todo_fields: dict[str, Any] | None = None, + capability_evidence: CapabilityReplanEvidence | None = None, + guard_capability: Mapping[str, Any] | None = None, ) -> RefreshReplanQualification: """Qualify one refresh's replan delta and accountable settlement outcome.""" @@ -495,6 +528,7 @@ def qualify_refresh_replan_writeback( ) semantic_delta = enforce_open_replan_writeback( + capability_evidence=capability_evidence, guard_capability=guard_capability, newest_first_runs=newest_first_runs, state_text=state_text, agent_id=agent_id, @@ -512,6 +546,8 @@ def qualify_refresh_replan_writeback( ), ) if semantic_delta: + if semantic_delta.get("retirement") and delivery_outcome != "outcome_gap": + raise ValueError("capability duty retirement requires outcome_gap; it is not delivery progress") effective_recorded = True classification = requested_classification delivery_outcome = requested_delivery_outcome diff --git a/loopx/orchestration.py b/loopx/orchestration.py index 76c060d5ba..7bdaad2e31 100644 --- a/loopx/orchestration.py +++ b/loopx/orchestration.py @@ -128,6 +128,12 @@ def compact_explore_harness_policy(policy: Any) -> dict[str, Any]: profile = str(harness.get("profile") or "").strip() if profile: compact["profile"] = profile + if "composition_mode" in harness or "composition_scope_id" in harness: + from .capabilities.explore.research_evidence import _research_result + policy = _research_result("explore.research.composition_policy", {"harness": harness}) + compact["composition_mode"] = policy["mode"] + if policy["coverage_scope_id"] is not None: + compact["composition_scope_id"] = policy["coverage_scope_id"] return compact diff --git a/loopx/presentation/renderers/status_markdown.py b/loopx/presentation/renderers/status_markdown.py index 7ab021b9a5..5832d4f226 100644 --- a/loopx/presentation/renderers/status_markdown.py +++ b/loopx/presentation/renderers/status_markdown.py @@ -629,6 +629,15 @@ def append_project_asset_warning_markdown( f"trigger_count={replan_obligation.get('trigger_count')} " f"triggers={markdown_scalar(','.join(trigger_kinds))}" ) + research = as_dict(project_asset.get("bounded_research_frontier")) + if research: + lines.append(" - research_execution: " + f"state={markdown_scalar(research.get('state'))} " + f"pending={research.get('pending_count')} scheduled={research.get('scheduled_count')} " + f"observed={research.get('observed_count')} ineligible={research.get('ineligible_count')} " + f"dismissed={research.get('dismissed_count')} deferred={research.get('deferred_count')}") + for gap in research.get("gaps") or []: + lines.append(f" - {markdown_scalar(gap.get('gap_id'))}: {markdown_scalar(gap.get('status'))}") interface_budget_cadence = ( project_asset.get("interface_budget_cadence") if isinstance(project_asset.get("interface_budget_cadence"), dict) diff --git a/loopx/quota.py b/loopx/quota.py index eb90263190..ced4333a46 100644 --- a/loopx/quota.py +++ b/loopx/quota.py @@ -1533,7 +1533,13 @@ def spend_quota_slot( "turn_instance_id": identity.turn_instance_id, "settlement_identity": identity.as_dict(), "settlement_result": settlement_result_payload(spent_result), - "reason": "quota spend receipt replayed for the same settlement identity", + **({"settlement_progress": settlement_readback.progress} + if settlement_readback.progress.get("closeout_kind") == "capability_duty_retired_no_spend" else {}), + "reason": ( + "exact Turn closeout replayed without a quota debit" + if settlement_readback.progress.get("closeout_kind") == "capability_duty_retired_no_spend" else + "quota spend receipt replayed for the same settlement identity" + ), } prior_spend_run = settlement_readback.spend_run if prior_spend_run is not None: diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index 47e4b85cb4..c3d04febe9 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -629,9 +629,17 @@ "api": "load_registry", "classification": "codec_api" }, + { + "site": "loopx/cli_commands/explore.py::._projection_for::codec_read:load_registry#1", + "line": 271, + "column": 20, + "kind": "codec_read", + "api": "load_registry", + "classification": "codec_api" + }, { "site": "loopx/cli_commands/explore.py::.handle_explore_command::codec_read:load_registry#1", - "line": 466, + "line": 486, "column": 20, "kind": "codec_read", "api": "load_registry", @@ -735,7 +743,7 @@ }, { "site": "loopx/cli_commands/project_lifecycle_refresh_state.py::.handle_refresh_state_command::codec_read:load_registry#1", - "line": 560, + "line": 561, "column": 17, "kind": "codec_read", "api": "load_registry", @@ -743,7 +751,7 @@ }, { "site": "loopx/cli_commands/project_lifecycle_refresh_state.py::.handle_refresh_state_command::codec_read:load_registry#2", - "line": 639, + "line": 640, "column": 17, "kind": "codec_read", "api": "load_registry", @@ -855,7 +863,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#1", - "line": 309, + "line": 324, "column": 49, "kind": "codec_read", "api": "load_registry", @@ -863,7 +871,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#2", - "line": 325, + "line": 340, "column": 24, "kind": "codec_read", "api": "load_registry", @@ -871,7 +879,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#3", - "line": 333, + "line": 348, "column": 24, "kind": "codec_read", "api": "load_registry", @@ -879,7 +887,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#4", - "line": 513, + "line": 528, "column": 53, "kind": "codec_read", "api": "load_registry", @@ -887,7 +895,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#5", - "line": 634, + "line": 649, "column": 61, "kind": "codec_read", "api": "load_registry", @@ -895,7 +903,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#6", - "line": 696, + "line": 711, "column": 13, "kind": "codec_read", "api": "load_registry", @@ -903,7 +911,7 @@ }, { "site": "loopx/cli_commands/todo.py::.handle_todo_command::codec_read:load_registry#7", - "line": 738, + "line": 753, "column": 38, "kind": "codec_read", "api": "load_registry", @@ -927,7 +935,7 @@ }, { "site": "loopx/configure_goal.py::.configure_goal::codec_transaction:project_registry_transaction#1", - "line": 515, + "line": 517, "column": 14, "kind": "codec_transaction", "api": "project_registry_transaction", @@ -935,7 +943,7 @@ }, { "site": "loopx/configure_goal.py::.configure_goal::codec_read:load_project_registry#1", - "line": 725, + "line": 727, "column": 14, "kind": "codec_read", "api": "load_project_registry", diff --git a/loopx/state_refresh.py b/loopx/state_refresh.py index 83ebdb851c..b5a2c5784e 100644 --- a/loopx/state_refresh.py +++ b/loopx/state_refresh.py @@ -906,7 +906,7 @@ def refresh_state_run( # Only pure input validation runs before this transitional persistence lock. with (nullcontext() if dry_run else exclusive_run_index_lock( runtime_root / "goals" / safe_goal_id / "runs" / "index.jsonl", operation="refresh-state" - )): + )), ExitStack() as research_write_guard: settlement_identity = None settlement_result = None delivery_workspace_causality = None @@ -1007,6 +1007,14 @@ def refresh_state_run( project_override=project, state_file_override=state_file, ) + harness = ((registry_goal or {}).get("spawn_policy") or {}).get("explore_harness") or {} + capability_guard = (getattr(settlement_readback, "semantic_replan_guard", None) or {}).get("selected_capability_guard") + if (harness.get("enabled") is True and harness.get("composition_mode") == "explicit_only") or capability_guard is not None: + from .capabilities.explore.result_log import explore_result_log_path + research_write_guard.enter_context(exclusive_file_lock( + explore_result_log_path(runtime_root, safe_goal_id), + agent_id=normalized_agent_id or None, operation="research-writeback", + )) planning_source = load_refresh_planning_source( runtime_root, safe_goal_id, resolved_state_file, require_display=bool(next_action) ) @@ -1135,7 +1143,13 @@ def refresh_state_run( and settlement_readback.semantic_replan_guard is not None else {} ) + from .capabilities.explore.research_frontier import prepare_research_replan_evidence + capability_evidence = prepare_research_replan_evidence(runtime_root=runtime_root, goal_id=safe_goal_id, + agent_id=normalized_agent_id, registry_goal=registry_goal, state_text=state_text, + capability_guard=settlement_replan_guard.get("selected_capability_guard")) replan_qualification = qualify_refresh_replan_writeback( + capability_evidence=capability_evidence, + guard_capability=settlement_replan_guard.get("selected_capability_guard"), todo_fields=todo_fields, autonomous_replan_recorded=autonomous_replan_recorded, requested_delta_kinds=normalized_repair_delta_kinds, diff --git a/loopx/todos.py b/loopx/todos.py index 43da08364a..c673410284 100644 --- a/loopx/todos.py +++ b/loopx/todos.py @@ -1710,6 +1710,12 @@ def complete_goal_todo( ) ) completion_state = completion_transaction.get("completion_state") + from .capabilities.explore.research_frontier import hold_research_completion_evidence + research_evidence = lease_fence_stack.enter_context(hold_research_completion_evidence( + registry_path=registry_path, runtime_root=shadow_runtime_root, goal_id=goal_id, + todo=completion_todo, state_text=original, + actor_agent_id=mutation_authority.get("actor_agent_id"), + )) completion_policy = completion_policy_from_transaction(completion_transaction) effective_claimed_by = completion_policy.effective_claimed_by registered_agents = completion_policy.registered_agents @@ -1826,6 +1832,7 @@ def complete_goal_todo( ) result = { "ok": True, + **({"capability_completion_evidence": research_evidence} if research_evidence is not None else {}), "dry_run": dry_run, "completed": True, "goal_id": goal_id, @@ -1926,6 +1933,11 @@ def supersede_goal_todo( idempotency_key=task_lease_idempotency_key, expected_version=task_lease_expected_version, runtime_root=shadow_runtime_root, ) + from .capabilities.explore.research_frontier import hold_research_completion_evidence + research_completion_evidence = lease_fence_stack.enter_context(hold_research_completion_evidence( + registry_path=registry_path, runtime_root=shadow_runtime_root, goal_id=goal_id, + todo=authority_todo, state_text=original, actor_agent_id=mutation_authority.get("actor_agent_id"), + )) update_result = apply_todo_update_to_lines( lines, todo_id=todo_id, @@ -1997,6 +2009,7 @@ def supersede_goal_todo( "ok": True, "dry_run": dry_run, "superseded": True, + **({"capability_completion_evidence": research_completion_evidence} if research_completion_evidence else {}), "goal_id": goal_id, **update_result, "changed": changed, From 1a5f6f413dd19afe60d76f4648a87a1856aeee40 Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Mon, 28 Sep 2026 21:36:39 -0700 Subject: [PATCH 02/12] test(explore): qualify composition write gates across real entrypoints Signed-off-by: Lihua <1017343802@qq.com> --- .../capability-scope.mjs | 51 ++- .../personal-workspace-browser/fixture.mjs | 18 +- .../test_explore_research_evidence.py | 107 +++++ .../test_research_composition_gate.py | 404 ++++++++++++++++++ .../test_goal_frontier_replan_rules.py | 21 + .../test_research_execution_authority.py | 278 ++++++++++++ .../blocked_hard_lease_lifecycle.test.ts | 21 + .../explore_research_execution.test.ts | 311 ++++++++++++++ .../quota_settlement_readback.test.ts | 72 ++++ .../control_plane_ts/replan_semantics.test.ts | 12 + tests/test_chat_goal_configuration_api.py | 64 +++ tsconfig.control-plane.json | 2 + 12 files changed, 1357 insertions(+), 4 deletions(-) create mode 100644 tests/capabilities/test_research_composition_gate.py create mode 100644 tests/control_plane/test_research_execution_authority.py create mode 100644 tests/control_plane_ts/explore_research_execution.test.ts diff --git a/examples/personal-workspace-browser/capability-scope.mjs b/examples/personal-workspace-browser/capability-scope.mjs index f90876ba6a..ed6f1bae19 100644 --- a/examples/personal-workspace-browser/capability-scope.mjs +++ b/examples/personal-workspace-browser/capability-scope.mjs @@ -50,6 +50,50 @@ export const capabilityScopeScenario = { await page.getByRole("button", { name: "预览变更", exact: true }).click(); await page.getByText("锁定 revision 的变更预览", { exact: true }).waitFor(); if (api.goalConfigurationRequests.at(-1)?.goal_id !== "research-monitor") throw new Error("Changed Goal was not used by preview"); + await catalog.getByRole("button", { name: /探索 Harness/ }).click(); + await page.getByRole("switch", { name: /^启用/u }).check(); + const composition = page.getByRole("combobox", { name: /^组合策略/u }); + await composition.selectOption("explicit_only"); + await page.getByLabel(/^研究覆盖范围/u).fill("joint-scope"); + await page.getByRole("button", { name: "预览变更", exact: true }).click(); + await page.getByText("锁定 revision 的变更预览", { exact: true }).waitFor(); + const preview = api.goalConfigurationRequests.at(-1); + if (preview?.goal_id !== "research-monitor" || preview?.configuration?.composition_mode !== "explicit_only" + || preview?.configuration?.composition_scope_id !== "joint-scope" || preview?.configuration?.enabled !== true) throw new Error("Composition preview lost its Goal, activation or typed scope"); + await composition.selectOption("disabled"); + if (await page.locator(".personal-capability-preview").count()) throw new Error("Composition edit retained a stale preview"); + await composition.selectOption("explicit_only"); + await page.getByRole("button", { name: "预览变更", exact: true }).click(); + await page.getByText("锁定 revision 的变更预览", { exact: true }).waitFor(); + await Promise.all([ + page.waitForResponse((response) => new URL(response.url()).pathname === "/api/chat/goal-configuration" && response.request().method() === "GET"), + page.getByRole("button", { name: "应用此预览", exact: true }).click(), + ]); + await catalog.getByRole("button", { name: /探索 Harness/ }).click(); + if (await composition.inputValue() !== "explicit_only" || await page.getByLabel(/^研究覆盖范围/u).inputValue() !== "joint-scope" + || !await page.getByRole("switch", { name: /^启用/u }).isChecked()) throw new Error("Applied composition configuration did not read back from its Goal"); + await composition.selectOption("disabled"); + await page.getByRole("button", { name: "预览变更", exact: true }).click(); + await page.getByText("锁定 revision 的变更预览", { exact: true }).waitFor(); + await Promise.all([ + page.waitForResponse((response) => new URL(response.url()).pathname === "/api/chat/goal-configuration" && response.request().method() === "GET"), + page.getByRole("button", { name: "应用此预览", exact: true }).click(), + ]); + await catalog.getByRole("button", { name: /探索 Harness/ }).click(); + if (await composition.inputValue() !== "disabled" || await page.getByLabel(/^研究覆盖范围/u).inputValue() !== "joint-scope" + || !await page.getByRole("switch", { name: /^启用/u }).isChecked()) throw new Error("Disabling composition changed its Goal activation or scope"); + const applied = api.goalConfigurationRequests.filter((request) => request.phase === "apply"); + if (applied.length !== 2 || applied.some((request) => request.goal_id !== "research-monitor" + || request.capability_id !== "explore_harness" || request.expected_plan_revision !== "sha256:goal-plan-explore_harness") + || applied[0].configuration.composition_mode !== "explicit_only" + || applied[1].configuration.composition_mode !== "disabled") throw new Error("Composition apply lost its locked revision or target"); + await page.screenshot({ path: resolve(outputDir, "capability-composition-desktop.png"), animations: "disabled" }); + await page.setViewportSize({ width: 390, height: 844 }); + await page.getByLabel(/^研究覆盖范围/u).scrollIntoViewIfNeeded(); + await page.getByLabel(/^研究覆盖范围/u).focus(); + await page.screenshot({ path: resolve(outputDir, "capability-composition-mobile.png"), animations: "disabled" }); + if (await page.evaluate(() => document.documentElement.scrollWidth > innerWidth + 1)) throw new Error("Composition controls overflow the narrow viewport"); + await page.setViewportSize({ width: 1512, height: 982 }); await defaults.check(); await page.getByRole("navigation", { name: "机器能力目录" }).waitFor(); if (await page.locator(".personal-capability-preview").count()) throw new Error("A Goal preview crossed into device defaults"); @@ -57,7 +101,7 @@ export const capabilityScopeScenario = { await page.getByRole("heading", { level: 2, name: "周期报告", exact: true }).waitFor(); if (await page.locator(".personal-capability-preview").count()) throw new Error("Returning revived a discarded preview"); if (!reads.includes("product-release") || !reads.includes("research-monitor")) throw new Error("Target changes did not read their own configuration"); - if (api.goalConfigurationRequests.some((r) => r.phase === "apply") || api.machineConfigurationRequests.some((r) => r.phase === "apply")) throw new Error("Navigation wrote configuration"); + if (api.goalConfigurationRequests.filter((r) => r.phase === "apply").length !== 2 || api.machineConfigurationRequests.some((r) => r.phase === "apply")) throw new Error("Navigation wrote configuration"); await page.screenshot({ path: resolve(outputDir, "capability-scope-goal.png"), animations: "disabled" }); await page.setViewportSize({ width: 390, height: 844 }); await target.focus(); @@ -70,6 +114,9 @@ export const capabilityScopeScenario = { if (!await goalScope.isChecked() || await target.inputValue() !== "product-release") throw new Error("Goal settings entry lost its target"); if (context.errors.length) throw new Error(context.errors.join(" | ")); return { coverageEntries: await context.close(), note: "One capability destination; explicit device/Goal scope; fresh target reads; no cross-scope drafts or writes; Goal entry preserved; desktop/mobile verified." }; - } catch (error) { await context.close(); throw error; } + } catch (error) { + await page.screenshot({ path: resolve(outputDir, "capability-scope-failure.png"), animations: "disabled" }); + await context.close(); throw error; + } }, }; diff --git a/examples/personal-workspace-browser/fixture.mjs b/examples/personal-workspace-browser/fixture.mjs index ccaad4d719..711d742572 100644 --- a/examples/personal-workspace-browser/fixture.mjs +++ b/examples/personal-workspace-browser/fixture.mjs @@ -172,6 +172,8 @@ export function goalCapabilityCatalog(multiSubagentConfiguration) { fields: [ { key: "enabled", label: "Enabled", description: "", input_kind: "boolean", required: false }, { key: "profile", label: "Planner profile", description: "", input_kind: "select", required: false, options: ["generic"] }, + { key: "composition_mode", label: "Composition policy", description: "Replan accepts an exact experiment successor or typed result.", input_kind: "select", required: false, options: ["disabled", "explicit_only"] }, + { key: "composition_scope_id", label: "Research coverage scope", description: "An opaque coverage scope.", input_kind: "text", required: false, nullable: true }, ], }), goalCapability({ capabilityId: "lark_kanban_heartbeat_sync", displayName: "Lark Kanban heartbeat sync" }), @@ -368,7 +370,7 @@ function filterStatusFixtureToScope(fixture, matchesScope) { export async function installApi(page, { goalSubagentConfigurationEnabled = true, initialActionProposals = [], managerChannelBinding = null, progressiveWorkspace = false, runtimeAgents = null } = {}) { let turnCounter = 0; - const runtime = page.__loopxRuntime ??= { actionProposals: new Map(), goalSubagentConfigurations: new Map(), larkConnections: [], messages: new Map(), sessions: new Map(), turnMessages: new Map() }; + const runtime = page.__loopxRuntime ??= { actionProposals: new Map(), goalSubagentConfigurations: new Map(), goalExploreConfigurations: new Map(), larkConnections: [], messages: new Map(), sessions: new Map(), turnMessages: new Map() }; const actionProposals = runtime.actionProposals; const sessions = runtime.sessions; const messages = runtime.messages; @@ -1224,7 +1226,8 @@ export async function installApi(page, { goalSubagentConfigurationEnabled = true machine_default_present: true, effective_revision: "sha256:periodic-effective", }, - }) : capability)], + }) : capability.capability_id === "explore_harness" && runtime.goalExploreConfigurations.has(goalId) + ? { ...capability, current: runtime.goalExploreConfigurations.get(goalId) } : capability)], }, }, status: 200 }); return; @@ -1245,6 +1248,17 @@ export async function installApi(page, { goalSubagentConfigurationEnabled = true if (url.pathname === "/api/chat/goal-configuration/apply" && request.method() === "POST") { const body = request.postDataJSON(); state.goalConfigurationRequests.push({ phase: "apply", ...body }); + if (body.capability_id === "explore_harness") { + runtime.goalExploreConfigurations.set(body.goal_id, body.configuration); + await route.fulfill({ contentType: "application/json", json: { + ok: true, schema_version: "goal_configuration_transaction_v0", status: "applied", + goal_id: body.goal_id, capability_id: body.capability_id, + plan_revision: body.expected_plan_revision, applied_revision: "sha256:goal-explore-applied", + readback_verified: true, changed_fields: ["explore_harness"], goal_configuration: body.configuration, + capability_catalog: { schema_version: "capability_configuration_catalog_v0", capabilities: [] }, + }, status: 200 }); + return; + } if (body.capability_id === "multi_subagent") { runtime.goalSubagentConfigurations.set(body.goal_id, body.configuration.enabled ? { mode: "multi_subagent", diff --git a/tests/capabilities/test_explore_research_evidence.py b/tests/capabilities/test_explore_research_evidence.py index e1d323b359..60373e2c99 100644 --- a/tests/capabilities/test_explore_research_evidence.py +++ b/tests/capabilities/test_explore_research_evidence.py @@ -13,6 +13,7 @@ build_explore_result_projection, explore_result_log_path, load_explore_result_events_strict, ) from loopx.extensions.lark.presentation.explore_results import _node_record_values +from loopx.todos import add_goal_todo, update_goal_todo GOAL = "research-fixture" @@ -176,3 +177,109 @@ def command(*args: str) -> dict: view = command("summary", "--goal-id", GOAL) assert view["nodes"][0]["research_observation"] == result["observation"] assert view["research_frontier"]["mode"] == "read_only_shadow" + + +def execution_fixture(tmp_path: Path) -> tuple[Path, Path, Path, dict]: + project = tmp_path / "project" + project.mkdir() + state = project / "ACTIVE_GOAL_STATE.md" + state.write_text(f"---\ngoal_id: {GOAL}\n---\n\n## Agent Todo\n\n") + runtime = tmp_path / "runtime" + registry = tmp_path / "registry.json" + registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{ + "id": GOAL, "repo": str(project), "state_file": state.name, "status": "active", + "coordination": {"agent_model": "peer_v1", "registered_agents": ["fixture-agent"]}, + }]})) + path = explore_result_log_path(runtime, GOAL) + for name in ["a", "b"]: + node(path, name) + append_research_observation(path, goal_id=GOAL, observation=observation("b")) + append_research_observation(path, goal_id=GOAL, observation=observation("a", target="b")) + node(path, "joint", kind="experiment") + for name in ["a", "b"]: + append_explore_result_event(path, build_explore_edge_event( + goal_id=GOAL, from_node="joint", to_node=name, edge_type="depends_on")) + todo = add_goal_todo( + registry_path=registry, goal_id=GOAL, role="agent", text="Run the bounded joint experiment.", + task_class="advancement_task", action_kind="joint_probe", claimed_by="fixture-agent", + explore_result_node_refs=["joint"], monitor_metadata={"target_key": "joint"}, + replan_obligation_id="replan-0123456789abcdef", + ) + gap = projection(path)["research_frontier"]["gaps"][0] + raw = observation("joint") + raw["progress"]["work_item_id"] = todo["todo_id"] + raw["input_observations"] = gap["input_observations"] + raw["execution_lineage"] = { + "schema_version": "research_execution_lineage_v0", "goal_id": GOAL, + "gap_id": gap["gap_id"], "replan_obligation_id": "replan-0123456789abcdef", + "successor_todo_id": todo["todo_id"], "agent_id": "fixture-agent", + } + return registry, runtime, path, raw + + +def test_real_execution_writer_reads_todo_and_replay_does_not_rebind(tmp_path: Path) -> None: + registry, runtime, path, raw = execution_fixture(tmp_path) + update_goal_todo( + registry_path=registry, goal_id=GOAL, todo_id=raw["execution_lineage"]["successor_todo_id"], + agent_id="fixture-agent", status="deferred", reason="await-synthetic-capacity", + resume_when="capacity_available:fixture", + ) + before = path.read_bytes() + with pytest.raises(ValueError, match="runnable joint-probe"): + append_research_observation(path, goal_id=GOAL, observation=raw, agent_id="fixture-agent", + registry_path=registry, runtime_root=runtime) + assert path.read_bytes() == before + update_goal_todo( + registry_path=registry, goal_id=GOAL, todo_id=raw["execution_lineage"]["successor_todo_id"], + agent_id="fixture-agent", status="open", clear_resume_when=True, + ) + result = append_research_observation(path, goal_id=GOAL, observation=raw, agent_id="fixture-agent", + registry_path=registry, runtime_root=runtime) + assert result["written"] + assert result["observation"]["execution_lineage"] == raw["execution_lineage"] + # A historical receipt remains read-only if current task or input authority + # changes. It cannot be refreshed into a new execution claim. + update_goal_todo( + registry_path=registry, goal_id=GOAL, todo_id=raw["execution_lineage"]["successor_todo_id"], + agent_id="fixture-agent", action_kind="inspect", + ) + node(path, "a", status="open") + before = path.read_bytes() + assert append_research_observation(path, goal_id=GOAL, observation=raw)["replayed"] + assert path.read_bytes() == before + assert projection(path)["research_frontier"]["observed_count"] == 0 + + +@pytest.mark.parametrize("batch", [False, True]) +def test_generic_writers_cannot_forge_new_execution_lineage(tmp_path: Path, batch: bool) -> None: + _registry, _runtime, path, raw = execution_fixture(tmp_path) + event = build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="resolved", research_observation=raw) + before = path.read_bytes() + with pytest.raises(ValueError, match="requires explore observe"): + if batch: + append_explore_result_events(path, [event], expected_goal_id=GOAL) + else: + append_explore_result_event(path, event) + assert path.read_bytes() == before + + +def test_cli_execution_requires_actor_and_preserves_lineage_in_readback(tmp_path: Path) -> None: + registry, runtime, path, raw = execution_fixture(tmp_path) + packet = tmp_path / "execution.json" + packet.write_text(json.dumps(raw)) + command = [sys.executable, "-m", "loopx.cli", "--registry", str(registry), "--format", "json", + "explore", "observe", "--goal-id", GOAL, "--observation-json", str(packet)] + before = path.read_bytes() + missing = subprocess.run(command, capture_output=True, text=True, check=False) + assert missing.returncode != 0 + assert "actor differs" in missing.stdout + assert path.read_bytes() == before + accepted = subprocess.run([*command, "--agent-id", "fixture-agent"], capture_output=True, text=True, check=False) + assert accepted.returncode == 0, accepted.stdout + accepted.stderr + result = json.loads(accepted.stdout) + assert result["observation"]["execution_lineage"] == raw["execution_lineage"] + readback = projection(path) + joint = next(row for row in readback["nodes"] if row["node_id"] == "joint") + assert joint["research_observation"] == result["observation"] + assert readback["research_frontier"]["observed_count"] == 1 diff --git a/tests/capabilities/test_research_composition_gate.py b/tests/capabilities/test_research_composition_gate.py new file mode 100644 index 0000000000..bbfa7a4379 --- /dev/null +++ b/tests/capabilities/test_research_composition_gate.py @@ -0,0 +1,404 @@ +"""Real CLI admission, successor and writeback paths with synthetic evidence.""" +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +from loopx.capabilities.explore.research_evidence import append_research_observation +from loopx.capabilities.explore.result_log import ( + append_explore_result_event, build_explore_edge_event, build_explore_node_event, explore_result_log_path, +) +from loopx.control_plane.work_items.semantic_replan_writeback import qualify_replan_writeback +from loopx.capabilities.explore.research_frontier import prepare_research_replan_evidence + +GOAL = "research-gate-fixture" +AGENT = "fixture-agent" + + +def observation(node: str, target: str | None = None) -> dict: + return {"schema_version": "typed_research_observation_v0", "explore_node_id": node, + "progress": {"schema_version": "typed_progress_observation_v0", "work_item_id": f"todo_{node}", + "result_class": "exploration_exhausted", "coverage_scope_id": f"scope-{node}", + "coverage_complete": True, "evidence_ids": [f"ev-{node}"]}, + "closure_basis": {"schema_version": "research_closure_basis_v0", "disposition": "bounded", + "constraints": [{"kind": "invariant", "id": "boundary", "role": "decisive"}], + "evidence_ids": [f"ev-{node}"]}, + "composition_candidates": [{"basis": "explicit", "target_node_id": target, + "interaction_kind": "state_interference", "evidence_ids": [f"ev-{node}", f"ev-{target}"]}] if target else []} + + +@pytest.fixture +def fixture(tmp_path: Path): + state = tmp_path / "ACTIVE_GOAL_STATE.md" + state.write_text('---\nstatus: active\nowner_mode: goal\nobjective: "Test the explicit research boundary."\n' + 'updated_at: 2026-09-28T00:00:00Z\n---\n\n# Research fixture\n\n' + '## Objective\n\nTest the explicit research boundary.\n\n## Next Action\n\nInspect the current evidence.\n\n' + '## Agent Todo\n\n- [ ] Observe the synthetic fixture.\n' + ' \n') + runtime, registry = tmp_path / "runtime", tmp_path / "registry.json" + registry.write_text(json.dumps({"schema_version": "0.1", "common_runtime_root": str(runtime), "goals": [{ + "id": GOAL, "repo": str(tmp_path), "state_file": state.name, "status": "active", "domain": "research-fixture", + "adapter": {"kind": "fixture_connected_delivery_v0", "status": "connected-delivery"}, "authority_sources": [], + "quota": {"compute": 1.0, "window_hours": 24, "allowed_slots": 20}, + "coordination": {"agent_model": "peer_v1", "registered_agents": [AGENT]}, + "spawn_policy": {"explore_harness": {"enabled": True}}, + }]})) + log = explore_result_log_path(runtime, GOAL) + for name in ["a", "b", "joint"]: + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id=name, title=f"Research {name}", + node_kind="experiment" if name == "joint" else "hypothesis", status="open" if name == "joint" else "resolved")) + append_research_observation(log, goal_id=GOAL, observation=observation("b")) + append_research_observation(log, goal_id=GOAL, observation=observation("a", "b")) + for name in ["a", "b"]: + append_explore_result_event(log, build_explore_edge_event(goal_id=GOAL, from_node="joint", to_node=name, edge_type="depends_on")) + + def cli(*args: str, success: bool = True) -> dict: + result = subprocess.run([sys.executable, "-m", "loopx.cli", "--format", "json", "--registry", str(registry), + "--runtime-root", str(runtime), *args], capture_output=True, text=True, check=False, + cwd=Path(__file__).resolve().parents[2]) + assert (result.returncode == 0) is success, result.stdout + result.stderr + return json.loads(result.stdout) + + return cli, log, runtime, registry, state + + +def activate(cli) -> None: + assert cli("configure-goal", "--goal-id", GOAL, "--explore-composition-mode", "explicit_only", + "--explore-composition-scope-id", "scope-joint", "--execute")["ok"] + + +def test_inactive_policy_never_requires_a_research_runtime(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr("loopx.capabilities.explore.composition_frontier.project_live_explore_composition_frontier", + lambda **_kwargs: pytest.fail("disabled policy must not read research state")) + for harness in [{"enabled": True}, {"enabled": True, "composition_mode": "disabled"}, + {"enabled": False, "composition_mode": "explicit_only", "composition_scope_id": "scope"}]: + obligation, _ = qualify_replan_writeback(newest_first_runs=[], state_text="## Agent Todo\n\n", agent_id=AGENT, + goal_id=GOAL, registry_goal={"id": GOAL, "spawn_policy": {"explore_harness": harness}}) + assert obligation is None + + +def test_real_guard_and_successor_share_the_current_gap(fixture) -> None: + cli, log, runtime, registry, state = fixture + old = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-off") + assert old.get("effective_action") != "autonomous_replan_required" + assert "research_execution_frontier" not in cli("explore", "summary", "--goal-id", GOAL) + off_status = cli("status", "--goal-id", GOAL, "--agent-id", AGENT) + assert all("bounded_research_frontier" not in item.get("project_asset", {}) + for item in off_status["attention_queue"]["items"]) + activate(cli) + guard = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-on") + assert guard["effective_action"] == "autonomous_replan_required" + packet = guard["replan_action_packet"] + obligation = packet["obligation_id"] + assert packet["capability_guard"]["gap_id"] == guard["bounded_research_frontier"]["selected_gap"]["gap_id"] + status = cli("status", "--goal-id", GOAL, "--agent-id", AGENT) + item = next(item for item in status["attention_queue"]["items"] if item["goal_id"] == GOAL) + assert item["project_asset"]["bounded_research_frontier"] == guard["bounded_research_frontier"] + assert item["project_asset"]["autonomous_replan_obligation"]["obligation_id"] == obligation + summary = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT) + assert summary["research_execution_frontier"] == guard["bounded_research_frontier"] + from loopx.cli_commands.explore import render_explore_markdown + from loopx.presentation.renderers.status_markdown import render_status_markdown + assert packet["capability_guard"]["gap_id"] in render_explore_markdown(summary) + assert packet["capability_guard"]["gap_id"] in render_status_markdown(status) + assert "lineage_gaps" not in guard["bounded_research_frontier"] + assert "settlement_transitions" not in guard["bounded_research_frontier"] + rejected = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", obligation, "--turn-instance-id", "fixture-on", + "--classification", "bounded_fixture_probe", "--delivery-batch-scale", "implementation", + "--delivery-outcome", "outcome_progress", "--progress-result-class", "advanced", + "--progress-surface-id", "unrelated", "--progress-evidence-id", "ev-unrelated", success=False) + assert "evidence duty" in json.dumps(rejected) + base = ["todo", "add", "--goal-id", GOAL, "--role", "agent", "--task-class", "advancement_task", + "--action-kind", "joint_probe", "--claimed-by", AGENT, "--replan-obligation-id", obligation] + bad = cli(*base, "--text", "Unrelated work.", "--explore-result-node-ref", "a", success=False) + assert "bind one current binary experiment" in json.dumps(bad) + deferred = cli(*base, "--text", "Deferred joint experiment.", "--explore-result-node-ref", "joint", + "--status", "deferred", "--resume-when", "capacity_available:fixture", success=False) + assert "no deferral" in json.dumps(deferred) + created = cli(*base, "--text", "Run the bounded joint experiment.", "--explore-result-node-ref", "joint", "--target-key", "joint") + assert created["replan_transition"]["recorded"] + before_closeout = state.read_bytes() + missing = cli("todo", "complete", "--goal-id", GOAL, "--todo-id", created["todo_id"], + "--agent-id", AGENT, "--claimed-by", AGENT, "--no-follow-up", + "--evidence", "No experiment result recorded.", success=False) + assert "current typed experiment observation" in json.dumps(missing) + assert state.read_bytes() == before_closeout + retired = cli("todo", "supersede", "--goal-id", GOAL, "--todo-id", created["todo_id"], + "--agent-id", AGENT, "--reason", "Attempt to close through another verb.", success=False) + assert "current typed experiment observation" in json.dumps(retired) + assert state.read_bytes() == before_closeout + replay = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-on") + assert replay["heartbeat_receipt"]["semantic_replan_capability_guard"] == packet["capability_guard"] + assert replay["replan_action_packet"]["obligation_id"] == obligation + refresh = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", obligation, "--turn-instance-id", "fixture-on", + "--classification", "bounded_fixture_probe", "--delivery-batch-scale", "implementation", + "--delivery-outcome", "outcome_progress", "--vision-summary", "Test this explicit research question.", + "--vision-acceptance", "The typed joint experiment addresses the question.", + "--vision-replan-trigger", "Research result remains open.") + assert refresh["ok"] + spent = cli("quota", "spend-slot", "--goal-id", GOAL, "--agent-id", AGENT, "--slots", "1", "--source", "heartbeat", + "--execute", "--replan-obligation-id", obligation, "--turn-instance-id", "fixture-on") + assert spent["settlement_progress"]["state"] == "settled" + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="resolved")) + result = observation("joint") + result["progress"]["work_item_id"] = created["todo_id"] + result["input_observations"] = guard["bounded_research_frontier"]["selected_gap"]["input_observations"] + result["execution_lineage"] = {"schema_version": "research_execution_lineage_v0", "goal_id": GOAL, + "gap_id": packet["capability_guard"]["gap_id"], "replan_obligation_id": obligation, + "successor_todo_id": created["todo_id"], "agent_id": AGENT} + append_research_observation(log, goal_id=GOAL, observation=result, agent_id=AGENT, + registry_path=registry, runtime_root=runtime) + complete = cli("todo", "complete", "--goal-id", GOAL, "--todo-id", created["todo_id"], + "--agent-id", AGENT, "--claimed-by", AGENT, "--no-follow-up", "--evidence", "Typed result recorded.") + assert complete["completed"] + assert complete["capability_completion_evidence"]["experiment_node_id"] == "joint" + archived = cli("todo", "archive-completed", "--goal-id", GOAL, "--role", "agent", + "--max-active-done", "0", "--execute") + assert archived["moved_count"] == 1 + assert cli("todo", "list", "--goal-id", GOAL, "--todo-id", created["todo_id"])["todos"][0]["archive_state"] == "archive" + from loopx.capabilities.explore.composition_frontier import project_live_explore_composition_frontier + frontier = project_live_explore_composition_frontier(runtime_root=runtime, goal_id=GOAL, agent_id=AGENT, + status_payload={"run_history": {"goals": json.loads(registry.read_text())["goals"]}}) + assert frontier["observed_count"] == 1 + assert frontier["scheduled_count"] == 0 + summary = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT) + assert summary["research_execution_frontier"]["observed_count"] == 1 + assert "lineage_gaps" not in summary["research_execution_frontier"] + from loopx.extensions.lark.presentation.explore_results import _node_record_values + joint = next(node for node in summary["nodes"] if node["node_id"] == "joint") + lark = _node_record_values(joint, goal_id=GOAL, source_id="synthetic-source") + assert "observed" in lark["Summary"] and packet["capability_guard"]["gap_id"] in lark["Summary"] + + +def test_selected_guard_cannot_be_erased_by_input_invalidation(fixture) -> None: + cli, log, _runtime, _registry, _state = fixture + activate(cli) + guard = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-stale") + obligation = guard["replan_action_packet"]["obligation_id"] + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="a", title="Research a", status="open")) + rejected = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", obligation, "--turn-instance-id", "fixture-stale", + "--classification", "bounded_fixture_probe", "--delivery-batch-scale", "implementation", + "--delivery-outcome", "outcome_progress", "--progress-result-class", "advanced", + "--progress-evidence-id", "ev-unrelated", success=False) + assert "evidence duty" in json.dumps(rejected) + + +def test_live_presentation_requires_explicit_registered_actor_for_multi_agent_goal(fixture) -> None: + cli, log, _runtime, registry, state = fixture + activate(cli) + source = json.loads(registry.read_text()) + source["goals"][0]["coordination"]["registered_agents"].append("other-agent") + registry.write_text(json.dumps(source)) + before = log.read_bytes(), state.read_bytes(), registry.read_bytes() + for args in [[], ["--agent-id", "unregistered"]]: + rejected = cli("explore", "summary", "--goal-id", GOAL, *args, success=False) + assert "registered Goal agent" in json.dumps(rejected) + summary = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT) + assert summary["research_execution_frontier"]["agent_id"] == AGENT + assert summary["research_execution_frontier"]["grants_execution_authority"] is False + assert (log.read_bytes(), state.read_bytes(), registry.read_bytes()) == before + + +def test_real_replan_guard_accepts_exact_blocker_wait_and_resumes_without_closure(fixture) -> None: + cli, log, runtime, registry, state = fixture + activate(cli) + guard = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-blocked") + packet = guard["replan_action_packet"] + task = cli("todo", "add", "--goal-id", GOAL, "--role", "agent", "--claimed-by", AGENT, + "--task-class", "advancement_task", "--action-kind", "joint_probe", "--target-key", "joint", + "--explore-result-node-ref", "joint", "--replan-obligation-id", packet["obligation_id"], + "--text", "Run the bounded joint experiment.") + blocker = cli("todo", "add", "--goal-id", GOAL, "--role", "agent", "--claimed-by", AGENT, + "--task-class", "blocker", "--action-kind", "investigate", "--unblocks-todo-id", task["todo_id"], + "--text", "Resolve the bounded dependency.") + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="blocked", blocked_reason="An identified dependency remains open.")) + result = observation("joint") + result["progress"].update(work_item_id=task["todo_id"], result_class="blocked", blocker_id=blocker["todo_id"]) + result["progress"].pop("coverage_complete") + result["closure_basis"] = None + result["input_observations"] = guard["bounded_research_frontier"]["selected_gap"]["input_observations"] + result["execution_lineage"] = {"schema_version": "research_execution_lineage_v0", "goal_id": GOAL, + "gap_id": packet["capability_guard"]["gap_id"], "replan_obligation_id": packet["obligation_id"], + "successor_todo_id": task["todo_id"], "agent_id": AGENT} + result["composition_resolution"] = {"schema_version": "research_composition_resolution_v0", + "disposition": "deferred", "evidence_ids": ["ev-joint"]} + recorded = append_research_observation(log, goal_id=GOAL, observation=result, agent_id=AGENT, + registry_path=registry, runtime_root=runtime) + cli("todo", "update", "--goal-id", GOAL, "--todo-id", task["todo_id"], "--agent-id", AGENT, + "--status", "blocked", "--resume-when", f"todo_done:{blocker['todo_id']}", "--reason", "Await the bounded dependency.") + frontier = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT)["research_execution_frontier"] + assert frontier["deferred_count"] == 1 and frontier["observed_count"] == 0 + refresh = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", packet["obligation_id"], "--turn-instance-id", "fixture-blocked", + "--classification", "bounded_fixture_blocker", "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_progress", + "--progress-work-item-id", task["todo_id"], "--progress-result-class", "blocked", + "--progress-blocker-id", blocker["todo_id"], "--progress-coverage-scope-id", "scope-joint", "--progress-evidence-id", "ev-joint", + "--vision-summary", "Test this explicit research question.", "--vision-acceptance", "The typed experiment addresses the question.", + "--vision-replan-trigger", "The experiment remains blocked with a typed resume condition.") + assert refresh["ok"] + saved = json.loads(Path(refresh["json_path"]).read_text()) + assert saved["autonomous_replan_ack"]["semantic_delta"]["capability_outcome"] == "composition_temporarily_deferred" + assert saved["progress_observation"] == recorded["observation"]["progress"] + spent = cli("quota", "spend-slot", "--goal-id", GOAL, "--agent-id", AGENT, "--slots", "1", "--source", "heartbeat", + "--execute", "--replan-obligation-id", packet["obligation_id"], "--turn-instance-id", "fixture-blocked") + assert spent["settlement_progress"]["state"] == "settled" + source = json.loads(registry.read_text())["goals"][0] + _, repeated = qualify_replan_writeback(newest_first_runs=[{"agent_id": AGENT, + "progress_observation": recorded["observation"]["progress"]}], state_text=state.read_text(), agent_id=AGENT, + goal_id=GOAL, registry_goal=source, guard_scoped=True, + capability_evidence=prepare_research_replan_evidence(runtime_root=runtime, goal_id=GOAL, + agent_id=AGENT, registry_goal=source, state_text=state.read_text(), capability_guard=packet["capability_guard"]), + guard_semantic_replan_obligation_id=packet["obligation_id"], guard_capability=packet["capability_guard"], + progress_observation=recorded["observation"]["progress"]) + assert repeated["accepted"] is False + cli("todo", "complete", "--goal-id", GOAL, "--todo-id", blocker["todo_id"], "--agent-id", AGENT, + "--no-follow-up", "--evidence", "Dependency resolved in the fixture.") + frontier = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT)["research_execution_frontier"] + assert frontier["deferred_count"] == 0 and frontier["pending_count"] == 1 + cli("todo", "update", "--goal-id", GOAL, "--todo-id", task["todo_id"], "--agent-id", AGENT, + "--status", "open", "--clear-resume-when", "--reason", "Dependency resolved; resume the experiment.") + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="open")) + frontier = cli("explore", "summary", "--goal-id", GOAL, "--agent-id", AGENT)["research_execution_frontier"] + assert frontier["scheduled_count"] == 1 and frontier["observed_count"] == 0 + + +@pytest.mark.parametrize("disposition", ["observed", "dismissed"]) +def test_real_replan_guard_accepts_exact_result_source_and_candidate_dismissal(fixture, disposition: str) -> None: + cli, log, runtime, registry, _state = fixture + activate(cli) + guard = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", "fixture-terminal") + packet = guard["replan_action_packet"] + task = cli("todo", "add", "--goal-id", GOAL, "--role", "agent", "--claimed-by", AGENT, + "--task-class", "advancement_task", "--action-kind", "joint_probe", "--target-key", "joint", + "--explore-result-node-ref", "joint", "--replan-obligation-id", packet["obligation_id"], + "--text", "Run the bounded joint experiment.") + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="joint", title="Research joint", + node_kind="experiment", status="dead_end" if disposition == "dismissed" else "resolved")) + result = observation("joint") + result["progress"]["work_item_id"] = task["todo_id"] + result["input_observations"] = guard["bounded_research_frontier"]["selected_gap"]["input_observations"] + result["execution_lineage"] = {"schema_version": "research_execution_lineage_v0", "goal_id": GOAL, + "gap_id": packet["capability_guard"]["gap_id"], "replan_obligation_id": packet["obligation_id"], + "successor_todo_id": task["todo_id"], "agent_id": AGENT} + if disposition == "dismissed": + result["progress"]["result_class"] = "no_followup" + result["closure_basis"]["disposition"] = "no_followup" + result["composition_resolution"] = {"schema_version": "research_composition_resolution_v0", + "disposition": "dismissed", "basis": "outside_scope", "evidence_ids": ["ev-joint"]} + recorded = append_research_observation(log, goal_id=GOAL, observation=result, agent_id=AGENT, + registry_path=registry, runtime_root=runtime) + refresh = cli("refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, + "--replan-obligation-id", packet["obligation_id"], "--turn-instance-id", "fixture-terminal", + "--classification", "bounded_fixture_terminal", "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_progress", + "--progress-work-item-id", task["todo_id"], "--progress-result-class", result["progress"]["result_class"], + "--progress-coverage-scope-id", "scope-joint", "--progress-coverage-complete", "--progress-evidence-id", "ev-joint", + "--vision-summary", "Test this explicit research question.", "--vision-acceptance", "The typed result addresses the question.", + "--vision-replan-trigger", "Retain the remaining Goal acceptance beyond this bounded candidate.") + saved = json.loads(Path(refresh["json_path"]).read_text()) + expected = "composition_candidate_dismissed" if disposition == "dismissed" else "composition_experiment_observed" + assert saved["autonomous_replan_ack"]["semantic_delta"]["capability_outcome"] == expected + assert saved["progress_observation"] == recorded["observation"]["progress"] + spent = cli("quota", "spend-slot", "--goal-id", GOAL, "--agent-id", AGENT, "--slots", "1", "--source", "heartbeat", + "--execute", "--replan-obligation-id", packet["obligation_id"], "--turn-instance-id", "fixture-terminal") + assert spent["settlement_progress"]["state"] == "settled" + complete = cli("todo", "supersede" if disposition == "dismissed" else "complete", "--goal-id", GOAL, + "--todo-id", task["todo_id"], "--agent-id", AGENT, + *( ["--reason", "Candidate dismissal is recorded."] if disposition == "dismissed" + else ["--evidence", "Typed scoped evidence is recorded.", "--no-follow-up"])) + assert complete["capability_completion_evidence"]["disposition"] == ( + "candidate_dismissed" if disposition == "dismissed" else "experiment_observed") + + +@pytest.mark.parametrize("change", ["input", "scope", "disabled"]) +def test_invalidated_original_turn_retires_with_no_spend_and_current_frontier_is_preserved(fixture, change: str) -> None: + cli, log, _runtime, _registry, state = fixture + activate(cli) + original = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--codex-app", + "--turn-instance-id", "fixture-invalidated") + packet = original["replan_action_packet"] + if change == "input": + append_explore_result_event(log, build_explore_node_event(goal_id=GOAL, node_id="a", title="Research a", status="open")) + elif change == "scope": + cli("configure-goal", "--goal-id", GOAL, "--explore-composition-scope-id", "replacement-scope", "--execute") + else: + cli("configure-goal", "--goal-id", GOAL, "--explore-composition-mode", "disabled", "--execute") + identity = ["--goal-id", GOAL, "--agent-id", AGENT, "--replan-obligation-id", packet["obligation_id"], + "--turn-instance-id", "fixture-invalidated"] + rejected = cli("refresh-state", *identity, "--classification", "fixture_retirement_probe", + "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_progress", + "--progress-result-class", "advanced", "--progress-evidence-id", "unrelated", success=False) + contract = rejected["replan_transition"]["retirement_contract"] + assert contract["blocking_todo_count"] == 0 + assert contract["original_guard"] == packet["capability_guard"] + progress = contract["progress_observation"] + # This is a causal lifecycle exit, not a way to count progress or debit. + invalid_progress = cli("refresh-state", *identity, "--classification", "fixture_false_progress", + "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_progress", + "--progress-result-class", "blocked", "--progress-blocker-id", progress["blocker_id"], + "--progress-evidence-id", progress["evidence_ids"][0], success=False) + assert "requires outcome_gap" in json.dumps(invalid_progress) + retired = cli("refresh-state", *identity, "--classification", "fixture_duty_invalidated", + "--delivery-batch-scale", "implementation", "--delivery-outcome", "outcome_gap", + "--progress-result-class", "blocked", "--progress-blocker-id", progress["blocker_id"], + "--progress-evidence-id", progress["evidence_ids"][0], + "--vision-summary", "Continue the bounded research question from current evidence.", + "--vision-acceptance", "Authoritative evidence must satisfy the research question.", + "--vision-replan-trigger", "Reassess the current frontier after this admitted basis changed.") + assert retired["settlement_progress"]["state"] == "settled" + assert retired["settlement_progress"]["closeout_kind"] == "capability_duty_retired_no_spend" + assert "settlement_owed" not in retired + denied = cli("quota", "spend-slot", *identity, "--slots", "1", "--source", "heartbeat", "--execute") + assert denied["appended"] is False and denied["idempotent_replay"] is True + assert denied["settlement_progress"]["closeout_kind"] == "capability_duty_retired_no_spend" + assert not any(receipt["step_kind"] == "quota_spend" for receipt in denied["settlement_result"]["receipts"]) + replay = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, + "--turn-instance-id", "fixture-invalidated", "--codex-app") + assert replay["heartbeat_receipt"]["semantic_replan_capability_guard"] == packet["capability_guard"] + assert replay["effective_action"] == "heartbeat_settled_skip" + fresh = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--codex-app", + "--turn-instance-id", "fixture-after-retirement") + assert fresh["effective_action"] != "unsettled_host_turn_recovery" + assert fresh["goal_frontier_projection"]["acceptance_gaps"] + assert "quota_slot_spent" not in json.dumps(retired) + assert "completed_at=" not in state.read_text() + + +def test_retirement_cannot_hide_a_runnable_bound_task_and_keeps_that_task_open(fixture) -> None: + cli, _log, _runtime, _registry, _state = fixture + activate(cli) + original = cli("quota", "should-run", "--goal-id", GOAL, "--agent-id", AGENT, "--codex-app", + "--turn-instance-id", "fixture-active-duty") + packet = original["replan_action_packet"] + task = cli("todo", "add", "--goal-id", GOAL, "--role", "agent", "--claimed-by", AGENT, + "--task-class", "advancement_task", "--action-kind", "joint_probe", "--target-key", "joint", + "--explore-result-node-ref", "joint", "--replan-obligation-id", packet["obligation_id"], "--text", "Run the bounded experiment.") + cli("configure-goal", "--goal-id", GOAL, "--explore-composition-scope-id", "replacement-scope", "--execute") + base = ["refresh-state", "--goal-id", GOAL, "--agent-id", AGENT, "--replan-obligation-id", packet["obligation_id"], + "--turn-instance-id", "fixture-active-duty", "--classification", "fixture_retirement", "--delivery-batch-scale", "implementation", + "--delivery-outcome", "outcome_gap", "--progress-result-class", "blocked"] + rejected = cli(*base, "--progress-blocker-id", "unrelated", "--progress-evidence-id", "unrelated", success=False) + contract = rejected["replan_transition"]["retirement_contract"] + assert contract["blocking_todo_ids"] == [task["todo_id"]] + progress = contract["progress_observation"] + cli(*base, "--progress-blocker-id", progress["blocker_id"], "--progress-evidence-id", progress["evidence_ids"][0], success=False) + assert cli("todo", "list", "--goal-id", GOAL, "--todo-id", task["todo_id"])["todos"][0]["status"] == "open" + # An explicit lifecycle pause keeps work and evidence visible; retirement + # still closes only the original duty, never the Todo or its Goal. + cli("todo", "update", "--goal-id", GOAL, "--todo-id", task["todo_id"], "--agent-id", AGENT, + "--status", "blocked", "--clear-resume-when", "--reason", "The admitted basis changed; reassess before execution.") + retired = cli(*base, "--progress-blocker-id", progress["blocker_id"], "--progress-evidence-id", progress["evidence_ids"][0], + "--vision-summary", "Reassess the bounded research question.", "--vision-acceptance", "Current evidence must satisfy the question.", + "--vision-replan-trigger", "Keep the acceptance and paused work visible.") + assert retired["settlement_progress"]["closeout_kind"] == "capability_duty_retired_no_spend" + assert cli("todo", "list", "--goal-id", GOAL, "--todo-id", task["todo_id"])["todos"][0]["status"] == "blocked" diff --git a/tests/control_plane/test_goal_frontier_replan_rules.py b/tests/control_plane/test_goal_frontier_replan_rules.py index 18b04d3d3f..65a32cca1c 100644 --- a/tests/control_plane/test_goal_frontier_replan_rules.py +++ b/tests/control_plane/test_goal_frontier_replan_rules.py @@ -159,6 +159,27 @@ def test_goal_frontier_replan_decision_table( ) +def test_capability_evidence_gap_precedence_and_disabled_parity() -> None: + # A caller-owned gap enters before monitor fallback. Existing authority, + # runnable work and acceptance checkpoints retain their established order. + selected = select_goal_frontier_replan_rule(GoalFrontierReplanFacts(capability_gap_pending=True)) + assert selected.rule is GoalFrontierReplanRule.CAPABILITY_EVIDENCE_GAP + assert selected.derives_obligation + for fields, expected in [ + ({"existing_replan_required": True}, GoalFrontierReplanRule.EXISTING_OBLIGATION), + ({"blocking_handoff_gate_count": 1}, GoalFrontierReplanRule.BLOCKING_HANDOFF_GATE), + ({"ready_deferred_successor_count": 1}, GoalFrontierReplanRule.READY_DEFERRED_SUCCESSOR), + ({"blocking_user_open_count": 1}, GoalFrontierReplanRule.OPEN_USER_TODO), + ({"acceptance_gap_count": 1}, GoalFrontierReplanRule.VISION_ACCEPTANCE_GAP), + ({"long_todo_chain_triggered": True}, GoalFrontierReplanRule.LONG_TODO_CHAIN), + ({"current_agent_blocker_count": 1}, GoalFrontierReplanRule.CURRENT_AGENT_BLOCKER), + ({"selectable_frontier_advancement": 1}, GoalFrontierReplanRule.NOT_MONITOR_ONLY), + ]: + decision = select_goal_frontier_replan_rule(GoalFrontierReplanFacts(capability_gap_pending=True, **fields)) + assert decision.rule is expected + assert select_goal_frontier_replan_rule(GoalFrontierReplanFacts()).rule is GoalFrontierReplanRule.NOT_MONITOR_ONLY + + def _repeat_vision_gap() -> list[dict[str, object]]: return [ { diff --git a/tests/control_plane/test_research_execution_authority.py b/tests/control_plane/test_research_execution_authority.py new file mode 100644 index 0000000000..b4c7cbed03 --- /dev/null +++ b/tests/control_plane/test_research_execution_authority.py @@ -0,0 +1,278 @@ +"""Execution evidence uses the real Todo provider, never a stale display row.""" +from __future__ import annotations + +import json +from pathlib import Path + +import pytest +from canonical_authority_fixture import initialize_canonical_authority, isolate_sqlite_runtime + +from loopx.capabilities.explore.research_evidence import append_research_observation +from loopx.capabilities.explore.result_log import ( + append_explore_result_event, build_explore_edge_event, build_explore_node_event, + build_explore_result_projection, explore_result_log_path, load_explore_result_events_strict, +) +from loopx.control_plane.coordination.runtime_shadow import build_todo_runtime_shadow_projection +from loopx.control_plane.effect_runtime import restart_effect_runtime +from loopx.todos import complete_goal_todo +from loopx.capabilities.explore.research_frontier import build_research_composition_frontier, prepare_research_replan_evidence + + +def observation(node: str) -> dict: + return { + "schema_version": "typed_research_observation_v0", "explore_node_id": node, + "progress": {"schema_version": "typed_progress_observation_v0", "work_item_id": f"todo_{node}", + "result_class": "exploration_exhausted", "coverage_scope_id": f"scope-{node}", + "coverage_complete": True, "evidence_ids": [f"ev-{node}"]}, + "closure_basis": {"schema_version": "research_closure_basis_v0", "disposition": "bounded", + "constraints": [{"kind": "invariant", "id": "boundary", "role": "decisive"}], + "evidence_ids": [f"ev-{node}"]}, + } + + +@pytest.mark.parametrize("provider", ["file", "sqlite"]) +@pytest.mark.parametrize("canonical_owner", ["fixture-agent", "other-agent"]) +def test_execution_attribution_reads_promoted_provider( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, canonical_owner: str, +) -> None: + isolate_sqlite_runtime(tmp_path, monkeypatch) + try: + goal = "research-provider-fixture" + state = tmp_path / "ACTIVE_GOAL_STATE.md" + state.write_text("# Goal\n\n## Agent Todo\n\n") + runtime = tmp_path / "runtime" + registry = tmp_path / "registry.json" + registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{ + "id": goal, "repo": str(tmp_path), "state_file": state.name, "status": "active", + "coordination": {"agent_model": "peer_v1", "registered_agents": ["fixture-agent", "other-agent"]}, + }]})) + task = { + "schema_version": "todo_item_v0", "todo_id": "todo_joint", "index": 1, + "role": "agent", "status": "open", "done": False, "text": "Run the bounded joint experiment.", + "archive_state": "active", "source_section": "Agent Todo", "priority": "P1", + "claimed_by": canonical_owner, "task_class": "advancement_task", "action_kind": "joint_probe", + "target_key": "joint", "explore_result_node_refs": ["joint"], + "replan_obligation_id": "replan-0123456789abcdef", + } + initialize_canonical_authority(runtime, goal, + build_todo_runtime_shadow_projection(goal_id=goal, todos=[task]), state_path=state, provider=provider) + # Even a plausible display claim is not the promoted source of truth. + state.write_text("# Goal\n\n## Agent Todo\n\n- [ ] Stale display\n" + " \n") + log = explore_result_log_path(runtime, goal) + for name in ["a", "b", "joint"]: + append_explore_result_event(log, build_explore_node_event( + goal_id=goal, node_id=name, title=f"Research {name}", status="resolved", + node_kind="experiment" if name == "joint" else "hypothesis")) + append_research_observation(log, goal_id=goal, observation=observation("b")) + source = observation("a") + source["composition_candidates"] = [{"basis": "explicit", "target_node_id": "b", + "interaction_kind": "state_interference", "evidence_ids": ["ev-a", "ev-b"]}] + append_research_observation(log, goal_id=goal, observation=source) + for name in ["a", "b"]: + append_explore_result_event(log, build_explore_edge_event( + goal_id=goal, from_node="joint", to_node=name, edge_type="depends_on")) + view = build_explore_result_projection(load_explore_result_events_strict(log, goal_id=goal), goal_id=goal) + gap = view["research_frontier"]["gaps"][0] + result = observation("joint") + result["input_observations"] = gap["input_observations"] + result["execution_lineage"] = { + "schema_version": "research_execution_lineage_v0", "goal_id": goal, "gap_id": gap["gap_id"], + "replan_obligation_id": task["replan_obligation_id"], "successor_todo_id": "todo_joint", "agent_id": "fixture-agent", + } + before = log.read_bytes(), state.read_bytes() + if canonical_owner == "fixture-agent": + receipt = append_research_observation(log, goal_id=goal, observation=result, agent_id="fixture-agent", + registry_path=registry, runtime_root=runtime) + assert receipt["written"] + else: + with pytest.raises(ValueError, match="same-agent runnable"): + append_research_observation(log, goal_id=goal, observation=result, agent_id="fixture-agent", + registry_path=registry, runtime_root=runtime) + assert log.read_bytes() == before[0] + assert state.read_bytes() == before[1] + finally: + restart_effect_runtime() + + +@pytest.mark.parametrize("provider", ["file", "sqlite"]) +@pytest.mark.parametrize("resolution", ["observed", "dismissed", "deferred"]) +def test_native_completion_refuses_missing_result_and_accepts_exact_observation( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, resolution: str, +) -> None: + isolate_sqlite_runtime(tmp_path, monkeypatch) + try: + goal, agent = "research-native-fixture", "fixture-agent" + state = tmp_path / "ACTIVE_GOAL_STATE.md" + state.write_text("# Goal\n\n## Agent Todo\n\n") + runtime, registry = tmp_path / "runtime", tmp_path / "registry.json" + harness = {"enabled": True, "composition_mode": "explicit_only", "composition_scope_id": "scope-joint"} + registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{ + "id": goal, "repo": str(tmp_path), "state_file": state.name, "status": "active", + "spawn_policy": {"explore_harness": harness}, + "coordination": {"agent_model": "peer_v1", "registered_agents": [agent]}, + }]})) + log = explore_result_log_path(runtime, goal) + for name in ["a", "b", "joint"]: + append_explore_result_event(log, build_explore_node_event( + goal_id=goal, node_id=name, title=f"Research {name}", + status="open" if name == "joint" else "resolved", + node_kind="experiment" if name == "joint" else "hypothesis")) + append_research_observation(log, goal_id=goal, observation=observation("b")) + source = observation("a") + source["composition_candidates"] = [{"basis": "explicit", "target_node_id": "b", + "interaction_kind": "state_interference", "evidence_ids": ["ev-a", "ev-b"]}] + append_research_observation(log, goal_id=goal, observation=source) + for name in ["a", "b"]: + append_explore_result_event(log, build_explore_edge_event( + goal_id=goal, from_node="joint", to_node=name, edge_type="depends_on")) + events = load_explore_result_events_strict(log, goal_id=goal) + projection = build_explore_result_projection(events, goal_id=goal) + frontier = build_research_composition_frontier(projection, + candidate_sources=[{"node_id": e["result_id"], "research_observation": e["research_observation"]} + for e in events if e.get("research_observation")], + harness=harness, todos=[], agent_id=agent) + gap = frontier["selected_gap"] + task = {"schema_version": "todo_item_v0", "todo_id": "todo_joint", "index": 1, + "role": "agent", "status": "open", "done": False, "text": "Run the bounded joint experiment.", + "archive_state": "active", "source_section": "Agent Todo", "priority": "P1", + "claimed_by": agent, "task_class": "advancement_task", "action_kind": "joint_probe", + "target_key": "joint", "explore_result_node_refs": ["joint"], "replan_obligation_id": gap["obligation_id"]} + initialize_canonical_authority(runtime, goal, + build_todo_runtime_shadow_projection(goal_id=goal, todos=[task]), state_path=state, provider=provider) + from loopx.control_plane.work_items.task_lease import acquire_task_lease + acquired = acquire_task_lease(registry_path=registry, runtime_root=runtime, goal_id=goal, + todo_id="todo_joint", owner=agent, idempotency_key="native-research-proof", ttl_seconds=300) + assert acquired["ok"] + from loopx.control_plane.coordination.local_authority import read_canonical_todos_if_promoted, LocalCoordinationAuthorityUnavailable + from loopx.control_plane.todos import provider_terminal_lifecycle as terminal_adapter + requests = [] + native_call = terminal_adapter.effect_runtime_result + def capture_native(method, params, **kwargs): + if method == "coordination.local_authority.todo_terminal": + requests.append(json.loads(json.dumps(params))) + return native_call(method, params, **kwargs) + monkeypatch.setattr(terminal_adapter, "effect_runtime_result", capture_native) + before = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + with pytest.raises(ValueError, match="current typed experiment observation"): + complete_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", + agent_id=agent, claimed_by=agent, no_followup=True, evidence="Synthetic result required.", + task_lease_idempotency_key="native-research-proof", task_lease_expected_version=acquired["lease"]["version"]) + forged = {**requests[-1], "operation_identity": {"kind": "explicit", "operation_id": "native-forged-approval"}, + "capability_completion_evidence": {"approved": True}} + direct = native_call("coordination.local_authority.todo_terminal", forged) + assert direct["changed"] is False + assert direct["reason_code"] == "research_experiment_result_required" + retired = native_call("coordination.local_authority.todo_terminal", {**forged, "command": "supersede", + "operation_identity": {"kind": "explicit", "operation_id": "native-supersede-without-result"}, + "requested_no_followup": False, "reason": "Attempt to close through another verb.", "completion_policy_request": None}) + assert retired["changed"] is False + assert retired["reason_code"] == "research_experiment_result_required" + after = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + assert after["provider_revision"] == before["provider_revision"] + assert after["todos"][0]["status"] == "open" + append_explore_result_event(log, build_explore_node_event(goal_id=goal, node_id="joint", + title="Research joint", node_kind="experiment", + status={"observed": "resolved", "dismissed": "dead_end", "deferred": "blocked"}[resolution], + blocked_reason="An identified dependency remains open." if resolution == "deferred" else None)) + result = observation("joint") + result["input_observations"] = gap["input_observations"] + result["execution_lineage"] = {"schema_version": "research_execution_lineage_v0", "goal_id": goal, + "gap_id": gap["gap_id"], "replan_obligation_id": gap["obligation_id"], "successor_todo_id": "todo_joint", "agent_id": agent} + if resolution == "dismissed": + result["progress"]["result_class"] = "no_followup" + result["closure_basis"]["disposition"] = "no_followup" + result["composition_resolution"] = {"schema_version": "research_composition_resolution_v0", + "disposition": "dismissed", "basis": "outside_scope", "evidence_ids": ["ev-joint"]} + elif resolution == "deferred": + from loopx.todos import add_goal_todo + blocker = add_goal_todo(registry_path=registry, goal_id=goal, role="agent", + text="Resolve the bounded dependency.", task_class="blocker", action_kind="investigate", + claimed_by=agent, agent_id=agent, unblocks_todo_id="todo_joint") + result["progress"].update(result_class="blocked", coverage_complete=False, blocker_id=blocker["todo_id"]) + result["closure_basis"] = None + result["composition_resolution"] = {"schema_version": "research_composition_resolution_v0", + "disposition": "deferred", "evidence_ids": ["ev-joint"]} + recorded = append_research_observation(log, goal_id=goal, observation=result, agent_id=agent, + registry_path=registry, runtime_root=runtime) + if resolution == "deferred": + from loopx.todos import update_goal_todo + from loopx.capabilities.explore.composition_frontier import project_live_explore_composition_frontier + from loopx.control_plane.work_items.semantic_replan_writeback import qualify_replan_writeback + from loopx.control_plane.work_items.task_lease import release_task_lease + with pytest.raises(LocalCoordinationAuthorityUnavailable, match="Release the active execution lease"): + update_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", agent_id=agent, + status="blocked", resume_when=f"todo_done:{blocker['todo_id']}", reason="Await the bounded dependency.") + released = release_task_lease(registry_path=registry, runtime_root=runtime, goal_id=goal, + todo_id="todo_joint", owner=agent, idempotency_key="native-research-proof", + expected_version=acquired["lease"]["version"]) + assert released["ok"] + update_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", agent_id=agent, + status="blocked", resume_when=f"todo_done:{blocker['todo_id']}", reason="Await the bounded dependency.") + source = json.loads(registry.read_text())["goals"][0] + def live_deferred(): + return project_live_explore_composition_frontier(runtime_root=runtime, goal_id=goal, + agent_id=agent, status_payload={"run_history": {"goals": [source]}}) + assert live_deferred()["deferred_count"] == 1 + _, delta = qualify_replan_writeback(newest_first_runs=[], state_text=state.read_text(), agent_id=agent, + goal_id=goal, registry_goal=source, guard_scoped=True, + capability_evidence=prepare_research_replan_evidence(runtime_root=runtime, goal_id=goal, + agent_id=agent, registry_goal=source, state_text=state.read_text()), + guard_semantic_replan_obligation_id=gap["obligation_id"], progress_observation=recorded["observation"]["progress"], + guard_capability={"capability_id": "explore", "gap_id": gap["gap_id"], "frontier_revision": gap["frontier_revision"]}) + assert delta["capability_outcome"] == "composition_temporarily_deferred" + revision = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal)["provider_revision"] + with pytest.raises(ValueError): + complete_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", agent_id=agent, + no_followup=True, task_lease_idempotency_key="native-research-proof", + task_lease_expected_version=acquired["lease"]["version"]) + assert read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal)["provider_revision"] == revision + dependency_lease = acquire_task_lease(registry_path=registry, runtime_root=runtime, goal_id=goal, + todo_id=blocker["todo_id"], owner=agent, idempotency_key="dependency-resolved", ttl_seconds=300) + complete_goal_todo(registry_path=registry, goal_id=goal, todo_id=blocker["todo_id"], agent_id=agent, + no_followup=True, evidence="Dependency resolved in the synthetic fixture.", + task_lease_idempotency_key="dependency-resolved", task_lease_expected_version=dependency_lease["lease"]["version"]) + assert live_deferred()["deferred_count"] == 0 + assert live_deferred()["pending_count"] == 1 + update_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", agent_id=agent, + status="open", clear_resume_when=True, reason="Dependency resolved; resume this experiment.") + append_explore_result_event(log, build_explore_node_event(goal_id=goal, node_id="joint", + title="Research joint", node_kind="experiment", status="open")) + assert live_deferred()["scheduled_count"] == 1 + assert live_deferred()["observed_count"] == 0 + reacquired = acquire_task_lease(registry_path=registry, runtime_root=runtime, goal_id=goal, + todo_id="todo_joint", owner=agent, idempotency_key="native-research-resumed", ttl_seconds=300) + assert reacquired["ok"] + assert reacquired["lease"]["version"] > acquired["lease"]["version"] + return + accepted = complete_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", + agent_id=agent, claimed_by=agent, clear_claim=True, no_followup=True, evidence="Typed synthetic result recorded.", + task_lease_idempotency_key="native-research-proof", task_lease_expected_version=acquired["lease"]["version"]) + assert accepted["completed"] + assert accepted["capability_completion_evidence"]["experiment_node_id"] == "joint" + completed = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + assert completed["todos"][0]["status"] == "done" + assert completed["todos"][0].get("claimed_by") is None + from loopx.control_plane.todos.provider_terminal_lifecycle import archive_canonical_todos_if_promoted + archived = archive_canonical_todos_if_promoted(registry_path=registry, runtime_root=runtime, + goal_id=goal, role="agent", max_active_done=0, dry_run=False) + assert archived["moved_todo_ids"] == ["todo_joint"] + completed = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + assert completed["todos"][0]["archive_state"] == "archive" + from loopx.capabilities.explore.composition_frontier import project_live_explore_composition_frontier + def live(): + return project_live_explore_composition_frontier(runtime_root=runtime, goal_id=goal, + agent_id=agent, status_payload={"run_history": {"goals": json.loads(registry.read_text())["goals"]}}) + assert live()[f"{resolution}_count"] == 1 + assert live()["scheduled_count"] == 0 + append_explore_result_event(log, build_explore_node_event(goal_id=goal, node_id="a", title="Research a", status="open")) + assert live()["observed_count"] == 0 + replay = complete_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", + agent_id=agent, claimed_by=agent, clear_claim=True, no_followup=True, evidence="Historical typed result remains immutable.", + task_lease_idempotency_key="native-research-proof", task_lease_expected_version=acquired["lease"]["version"]) + assert replay["provider_status"] == "replayed" + assert replay["capability_completion_evidence"] == accepted["capability_completion_evidence"] + latest = read_canonical_todos_if_promoted(runtime_root=runtime, goal_id=goal) + assert latest["provider_revision"] == completed["provider_revision"] + finally: + restart_effect_runtime() diff --git a/tests/control_plane_ts/blocked_hard_lease_lifecycle.test.ts b/tests/control_plane_ts/blocked_hard_lease_lifecycle.test.ts index 3ae99b0989..4244b47134 100644 --- a/tests/control_plane_ts/blocked_hard_lease_lifecycle.test.ts +++ b/tests/control_plane_ts/blocked_hard_lease_lifecycle.test.ts @@ -123,6 +123,27 @@ for (const provider of ["file", "sqlite"] as const) { assert.equal((after.head.leases as JsonObject[])[0]!.status, "released"); }); + providerTest(`${provider}: typed prerequisite wait preserves pause fencing and grants no execution`, async () => { + const store = await seeded(provider, "2026-09-25T05:00:00Z"); + const input = {...pause("pause-typed-wait"), planning_intent: { + status: "blocked", reason: "Await the identified prerequisite", resume_when: "todo_done:todo_successor"}}; + assert.equal((await executeCoordinationTodoUpdate(store, input)).status, "applied"); + const after = await read(store), todo = (after.head.todos as JsonObject[])[0]!; + assert.equal(todo.status, "blocked"); + assert.equal(todo.resume_when, "todo_done:todo_successor"); + assert.equal(todo.claimed_by, OWNER); + assert.equal((after.head.leases as JsonObject[])[0]!.status, "released"); + assert.equal((await executeCoordinationTodoUpdate(store, input)).status, "replayed"); + assert.deepEqual(await read(store), after); + const active = await seeded(provider, "2026-09-25T07:00:00Z"); + const before = await read(active); + assert.equal((await executeCoordinationTodoUpdate(active, input)).reason_code, "blocked_lifecycle_active_lease"); + assert.equal((await executeCoordinationTodoUpdate(active, {...input, + lease_idempotency_key: "old-execution", lease_expected_version: 29})).reason_code, + "blocked_lifecycle_execution_proof_not_allowed"); + assert.deepEqual(await read(active), before); + }); + providerTest(`${provider}: live lease, unauthorized actor and bundled work are rejected`, async () => { const active = await seeded(provider, "2026-09-25T07:00:00Z"); const before = await read(active); diff --git a/tests/control_plane_ts/explore_research_execution.test.ts b/tests/control_plane_ts/explore_research_execution.test.ts new file mode 100644 index 0000000000..591e6c6000 --- /dev/null +++ b/tests/control_plane_ts/explore_research_execution.test.ts @@ -0,0 +1,311 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import type {JsonObject} from "../../loopx/control_plane/effect_program.ts"; +import {normalizeResearchObservation, projectResearchFrontier, researchCompositionGaps} from "../../loopx/control_plane/capabilities/explore_research.ts"; +import {validateResearchExecution, normalizeResearchCompositionPolicy, projectResearchComposition, + researchCompositionFacts, qualifyResearchCompositionWriteback, + validateResearchCompositionSuccessor, qualifyResearchCompletion} from "../../loopx/control_plane/capabilities/explore_research_execution.ts"; + +function fixture(): JsonObject { + const observation = (node: string, target?: string): JsonObject => ({ + schema_version: "typed_research_observation_v0", explore_node_id: node, + progress: {schema_version: "typed_progress_observation_v0", work_item_id: `todo_${node}`, + result_class: "exploration_exhausted", coverage_scope_id: `scope-${node}`, + coverage_complete: true, evidence_ids: [`ev-${node}`]}, + closure_basis: {schema_version: "research_closure_basis_v0", disposition: "bounded", + constraints: [{kind: "invariant", id: "boundary", role: "decisive"}], evidence_ids: [`ev-${node}`]}, + composition_candidates: target ? [{target_node_id: target, basis: "explicit", + interaction_kind: "state_interference", evidence_ids: [`ev-${node}`, `ev-${target}`]}] : [], + }); + const nodes = ["a", "b"].map(node_id => ({node_id, node_kind: "hypothesis", status: "resolved", + research_observation: normalizeResearchObservation({observation: observation(node_id, node_id === "a" ? "b" : undefined)})})); + const params: JsonObject = {goal_id: "fixture", agent_id: "fixture-agent", nodes: [ + ...nodes, {node_id: "joint", node_kind: "experiment", status: "open", agent_id: "fixture-agent"}], + edges: ["a", "b"].map(to_node => ({from_node: "joint", to_node, edge_type: "depends_on"}))}; + const gap = researchCompositionGaps(params)[0]; + params.observation = {...observation("joint"), input_observations: gap.input_observations, + progress: {...observation("joint").progress as JsonObject, work_item_id: "todo_joint"}, + execution_lineage: {schema_version: "research_execution_lineage_v0", goal_id: "fixture", + gap_id: gap.gap_id, replan_obligation_id: "replan-0123456789abcdef", successor_todo_id: "todo_joint", agent_id: "fixture-agent"}}; + params.todo = {todo_id: "todo_joint", replan_obligation_id: "replan-0123456789abcdef", + claimed_by: "fixture-agent", task_class: "advancement_task", action_kind: "joint_probe", + explore_result_node_refs: ["joint"], target_key: "joint", actionable_open: true}; + return params; +} + +function liveFixture(): JsonObject { + const params = fixture(); + params.harness = {enabled: true, composition_mode: "explicit_only", composition_scope_id: "scope-joint"}; + params.todo = {...params.todo as JsonObject, status: "open"}; + const facts = researchCompositionFacts(params); + params.bindings = (facts.gaps as JsonObject[]).map(gap => ({gap_id: gap.gap_id, + obligation: {obligation_id: "replan-0123456789abcdef"}})); + params.todos = []; + const raw = params.observation as JsonObject; + params.observation = {...raw, progress: {...raw.progress as JsonObject, fingerprint: "progress-joint"}}; + return params; +} + +test("live policy is explicit, scoped and default off", () => { + assert.equal(normalizeResearchCompositionPolicy({harness: {enabled: true}}).enabled, false); + assert.equal(normalizeResearchCompositionPolicy({harness: {enabled: "true", composition_mode: "explicit_only", composition_scope_id: "scope"}}).enabled, false); + assert.throws(() => normalizeResearchCompositionPolicy({harness: {enabled: true, composition_mode: "explicit_only"}}), /scope/); + assert.throws(() => normalizeResearchCompositionPolicy({harness: {enabled: true, composition_mode: "infer"}}), /composition_mode/); + const params = liveFixture(); + params.harness = {enabled: true}; + assert.equal(projectResearchComposition(params).state, "disabled"); +}); + +test("only the exact runnable experiment successor schedules the live gap", () => { + const params = liveFixture(); + assert.equal(projectResearchComposition(params).pending_count, 1); + params.todos = [params.todo]; + const scheduled = projectResearchComposition(params); + assert.equal(scheduled.scheduled_count, 1); + assert.equal(scheduled.observed_count, 0); + for (const patch of [{claimed_by: "another-agent"}, {replan_obligation_id: "replan-fedcba9876543210"}, + {actionable_open: false, status: "deferred"}, {explore_result_node_refs: ["other"]}, {action_kind: "read"}]) { + assert.equal(projectResearchComposition({...params, todos: [{...params.todo as JsonObject, ...patch}]}).pending_count, 1); + } + const pending = projectResearchComposition({...params, todos: []}); + assert.equal(validateResearchCompositionSuccessor({frontier: pending, todo: params.todo, agent_id: params.agent_id}).accepted, true); + assert.throws(() => validateResearchCompositionSuccessor({frontier: pending, + todo: {...params.todo as JsonObject, status: "deferred", resume_when: "capacity_available:fixture"}, agent_id: params.agent_id}), /no deferral/); +}); + +test("live observation needs execution lineage and cannot be replaced by generic progress or a replay", () => { + const params = liveFixture(); + const pending = projectResearchComposition(params); + const gap = (pending.lineage_gaps as JsonObject[])[0]; + const guard = {schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore", + gap_id: gap.gap_id, frontier_revision: gap.frontier_revision}; + const qualify = (frontier: JsonObject, progress: JsonObject, repeated: string[] = []) => qualifyResearchCompositionWriteback({ + frontier, capability_guard: guard, obligation_id: "replan-0123456789abcdef", progress_observation: progress, + claimed_progress_fingerprints: repeated}); + assert.equal(qualify(pending, {fingerprint: "unrelated", result_class: "advanced"}).accepted, false); + const nodes = params.nodes as JsonObject[], raw = params.observation as JsonObject; + const unbound = {...raw}; delete unbound.execution_lineage; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", + research_observation: normalizeResearchObservation({observation: unbound})}]; + params.todos = [params.todo]; + assert.equal(projectResearchComposition(params).observed_count, 0); + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", + research_observation: normalizeResearchObservation({observation: raw})}]; + const observed = projectResearchComposition(params), progress = raw.progress as JsonObject; + assert.equal(observed.observed_count, 1); + assert.equal(qualify(observed, progress).capability_outcome, "composition_experiment_observed"); + assert.equal(qualify(observed, progress, ["progress-joint"]).accepted, false); + assert.equal(qualify(observed, {...progress, evidence_ids: []}).accepted, false); + assert.equal(qualify(observed, {...progress, work_item_id: "todo_other"}).accepted, false); + params.nodes = [{...nodes[0], status: "open"}, nodes[1], (params.nodes as JsonObject[])[2]]; + const invalidated = projectResearchComposition(params); + assert.equal(invalidated.ineligible_count, 1); + assert.equal(qualify(invalidated, progress).accepted, false); +}); + +test("Todo closeout needs this task's current terminal experiment result", () => { + const params = liveFixture(); + params.todos = [params.todo]; + const request = (frontier: JsonObject, todo = params.todo) => qualifyResearchCompletion({ + frontier, todo, actor_agent_id: params.agent_id, goal_id: params.goal_id}); + assert.equal(request(projectResearchComposition(params)).allowed, false); + const nodes = params.nodes as JsonObject[]; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", + research_observation: normalizeResearchObservation({observation: params.observation})}]; + const observed = projectResearchComposition(params); + const accepted = request(observed); + assert.equal(accepted.allowed, true); + assert.equal((accepted.evidence as JsonObject).todo_id, "todo_joint"); + assert.equal((accepted.evidence as JsonObject).experiment_node_id, "joint"); + assert.equal(request(observed, {...params.todo as JsonObject, todo_id: "todo_other"}).allowed, false); + params.nodes = [{...nodes[0], status: "open"}, nodes[1], (params.nodes as JsonObject[])[2]]; + assert.equal(request(projectResearchComposition(params)).allowed, false); + assert.equal(qualifyResearchCompletion({frontier: {enabled: false}, todo: params.todo}).required, false); +}); + +test("execution evidence binds Goal, gap, obligation, Todo, actor, experiment and current inputs", () => { + const params = fixture(); + const canonical = validateResearchExecution(params); + assert.deepEqual(canonical.execution_lineage, (params.observation as JsonObject).execution_lineage); + const unbound = {...params.observation as JsonObject}; + delete unbound.execution_lineage; + assert.notEqual(canonical.fingerprint, normalizeResearchObservation({observation: unbound}).fingerprint); + // The receipt and cold projection retain diagnostic semantics. No execution, + // scientific truth or Goal completion is inferred from validation alone. + assert.equal(projectResearchFrontier(params).mode, "read_only_shadow"); + assert.equal(researchCompositionGaps(params)[0].state, "pending"); +}); + +test("completed archive lineage retains evidence without making archived work runnable", () => { + const params = liveFixture(), nodes = params.nodes as JsonObject[]; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", + research_observation: normalizeResearchObservation({observation: params.observation})}]; + const retained = {...params.todo as JsonObject, status: "done", archive_state: "archive", actionable_open: false}; + for (const todo of [retained, {...retained, claimed_by: null}]) { + const frontier = projectResearchComposition({...params, todos: [todo]}); + assert.equal(frontier.observed_count, 1); + assert.equal(frontier.scheduled_count, 0); + assert.equal((frontier.lineage_gaps as JsonObject[])[0].observed_todo_id, "todo_joint"); + } + for (const patch of [{status: "open"}, {replan_obligation_id: "replan-fedcba9876543210"}, + {explore_result_node_refs: ["other"]}, {claimed_by: "other-agent"}]) { + assert.equal(projectResearchComposition({...params, todos: [{...retained, ...patch}]}).observed_count, 0); + } + assert.equal(projectResearchComposition({...params, todos: []}).observed_count, 0); + assert.equal(projectResearchComposition({...params, todos: [retained], + nodes: [{...nodes[0], status: "open"}, nodes[1], (params.nodes as JsonObject[])[2]]}).observed_count, 0); + assert.throws(() => validateResearchExecution({...params, todo: retained}), /runnable/); +}); + +test("evidence-backed candidate dismissal retires work without claiming an experiment outcome", () => { + const params = liveFixture(), nodes = params.nodes as JsonObject[]; + const raw = params.observation as JsonObject; + const dismissed = {...raw, progress: {...raw.progress as JsonObject, result_class: "no_followup"}, + closure_basis: {...raw.closure_basis as JsonObject, disposition: "no_followup"}, + composition_resolution: {schema_version: "research_composition_resolution_v0", disposition: "dismissed", + basis: "outside_scope", evidence_ids: ["ev-joint"]}}; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "dead_end", + research_observation: normalizeResearchObservation({observation: dismissed})}]; + params.todos = [params.todo]; + const frontier = projectResearchComposition(params), gap = (frontier.lineage_gaps as JsonObject[])[0]; + assert.equal(frontier.dismissed_count, 1); + assert.equal(frontier.observed_count, 0); + assert.equal(projectResearchFrontier(params).observed_count, 0); + const guard = {capability_id: "explore", gap_id: gap.gap_id, frontier_revision: gap.frontier_revision}; + const qualified = qualifyResearchCompositionWriteback({frontier, capability_guard: guard, + obligation_id: gap.obligation_id, progress_observation: dismissed.progress, claimed_progress_fingerprints: []}); + assert.equal(qualified.capability_outcome, "composition_candidate_dismissed"); + assert.equal(qualifyResearchCompletion({frontier, todo: params.todo, goal_id: params.goal_id, + actor_agent_id: params.agent_id}).allowed, true); + for (const patch of [{basis: "not_promising"}, {evidence_ids: []}, {evidence_ids: ["unrelated"]}]) { + assert.throws(() => normalizeResearchObservation({observation: {...dismissed, + composition_resolution: {...dismissed.composition_resolution, ...patch}}})); + } +}); + +test("temporary deferral needs a fresh exact blocker and the common resume contract", () => { + const params = liveFixture(), nodes = params.nodes as JsonObject[]; + const raw = params.observation as JsonObject; + const blocked = {...raw, progress: {...raw.progress as JsonObject, result_class: "blocked", + coverage_complete: false, blocker_id: "todo_dependency"}, closure_basis: null, + composition_resolution: {schema_version: "research_composition_resolution_v0", disposition: "deferred", + evidence_ids: ["ev-joint"]}}; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "blocked", + research_observation: normalizeResearchObservation({observation: blocked})}]; + const task = {...params.todo as JsonObject, status: "deferred", actionable_open: false, + resume_when: "todo_done:todo_dependency"}; + const blocker = {todo_id: "todo_dependency", role: "agent", task_class: "blocker", status: "open", + claimed_by: "fixture-agent", archive_state: "active", unblocks_todo_id: "todo_joint"}; + const deferred = projectResearchComposition({...params, todos: [task, blocker]}); + assert.equal(deferred.deferred_count, 1); + assert.equal(deferred.observed_count, 0); + assert.equal(deferred.pending_count, 0); + assert.equal(qualifyResearchCompletion({frontier: deferred, todo: task}).allowed, false); + const gap = (deferred.lineage_gaps as JsonObject[])[0]; + const request = {frontier: deferred, capability_guard: {capability_id: "explore", gap_id: gap.gap_id, + frontier_revision: gap.frontier_revision}, obligation_id: gap.obligation_id, + progress_observation: blocked.progress, claimed_progress_fingerprints: [], claimed_blocker_ids: []}; + assert.equal(qualifyResearchCompositionWriteback(request).capability_outcome, "composition_temporarily_deferred"); + assert.equal(qualifyResearchCompositionWriteback({...request, claimed_blocker_ids: ["todo_dependency"]}).accepted, false); + for (const patch of [{status: "done"}, {status: "blocked"}, {claimed_by: "other-agent"}, {task_class: "continuous_monitor"}, + {unblocks_todo_id: "todo_other"}, {archive_state: "archive"}]) { + assert.equal(projectResearchComposition({...params, todos: [task, {...blocker, ...patch}]}).deferred_count, 0); + } + for (const resume_when of ["capacity_available:fixture", "todo_done:todo_unknown", null]) { + assert.equal(projectResearchComposition({...params, todos: [{...task, resume_when}, blocker]}).deferred_count, 0); + } + // Resolving the prerequisite exposes the same open gap; it grants no lease + // and cannot turn the deferred declaration into an experimental outcome. + const resumed = projectResearchComposition({...params, todos: [task, {...blocker, status: "done"}]}); + assert.equal(resumed.pending_count, 1); + assert.equal(resumed.observed_count, 0); +}); + +test("a changed evidence duty needs explicit retirement facts and never closes live work", () => { + const params = liveFixture(), initial = projectResearchComposition(params); + const selected = (initial.lineage_gaps as JsonObject[])[0]; + const guard = {schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore", + gap_id: selected.gap_id, frontier_revision: selected.frontier_revision}; + const nodes = params.nodes as JsonObject[]; + const current = projectResearchComposition({...params, nodes: [{...nodes[0], status: "open"}, ...nodes.slice(1)]}); + const request = {frontier: current, capability_guard: guard, obligation_id: selected.obligation_id}; + const rejected = qualifyResearchCompositionWriteback(request); + assert.equal(rejected.accepted, false); + const contract = rejected.retirement_contract as JsonObject; + assert.equal(contract.disposition, "invalidated"); + const accepted = qualifyResearchCompositionWriteback({...request, + progress_observation: {...contract.progress_observation as JsonObject, fingerprint: "retirement-proof"}}); + assert.equal(accepted.capability_outcome, "composition_duty_invalidated"); + assert.deepEqual(accepted.outcomes, ["capability_duty_retired"]); + assert.equal(qualifyResearchCompositionWriteback({...request, frontier: initial, + progress_observation: contract.progress_observation}).accepted, false); + for (const patch of [{work_item_id: "replan-fedcba9876543210"}, {blocker_id: "unrelated"}, + {result_class: "advanced"}, {evidence_ids: ["unrelated"]}]) { + assert.equal(qualifyResearchCompositionWriteback({...request, + progress_observation: {...contract.progress_observation as JsonObject, ...patch}}).accepted, false); + } + const blocked = projectResearchComposition({...params, nodes: [{...nodes[0], status: "open"}, ...nodes.slice(1)], todos: [params.todo]}); + assert.equal(qualifyResearchCompositionWriteback({...request, frontier: blocked, + progress_observation: contract.progress_observation}).accepted, false); + assert.equal(qualifyResearchCompositionWriteback({...request, frontier: {schema_version: "unavailable"}, + progress_observation: contract.progress_observation}).retirement_contract, undefined); +}); + +test("rejected, deferred, wrong-agent and unrelated successors cannot attribute an execution result", () => { + for (const patch of [ + {actionable_open: false}, {claimed_by: "another-agent"}, {task_class: "continuous_monitor"}, + {action_kind: "inspect"}, {todo_id: "todo_other"}, {replan_obligation_id: "replan-fedcba9876543210"}, + {explore_result_node_refs: ["other"]}, {explore_result_node_refs: ["joint", "other"]}, + {target_key: "other"}, {archive_state: "archive"}, {excluded_agents: ["fixture-agent"]}, + ]) { + const params = fixture(); + params.todo = {...params.todo as JsonObject, ...patch}; + assert.throws(() => validateResearchExecution(params), /research execution/); + } + for (const patch of [{agent_id: "another-agent"}, {goal_id: "other-goal"}, {gap_id: "research-composition-0000000000000000"}]) { + const params = fixture(), raw = params.observation as JsonObject; + params.observation = {...raw, execution_lineage: {...raw.execution_lineage as JsonObject, ...patch}}; + assert.throws(() => validateResearchExecution(params), /research execution/); + } +}); + +test("stale fingerprints, reads, malformed lineage and already-observed inputs fail closed", () => { + const params = fixture(), raw = params.observation as JsonObject; + assert.throws(() => validateResearchExecution({...params, observation: {...raw, + input_observations: [{node_id: "a", fingerprint: "stale"}, ...(raw.input_observations as JsonObject[]).slice(1)]}}), /exact current/); + for (const progress of [{...raw.progress as JsonObject, result_class: "unchanged"}, + {...raw.progress as JsonObject, result_class: "advanced", evidence_ids: []}]) { + assert.throws(() => validateResearchExecution({...params, observation: {...raw, progress, closure_basis: null}}), /read or ACK/); + } + assert.throws(() => normalizeResearchObservation({observation: {...raw, + execution_lineage: {...raw.execution_lineage as JsonObject, successor_todo_id: "todo_other"}}}), /exact gap/); + const nodes = params.nodes as JsonObject[]; + params.nodes = [...nodes.slice(0, 2), {...nodes[2], status: "resolved", research_observation: normalizeResearchObservation({observation: raw})}]; + assert.equal(researchCompositionGaps(params)[0].state, "observed"); + assert.throws(() => validateResearchExecution(params), /pending gap/); +}); + +test("the three-card presentation budget cannot hide execution attribution", () => { + const params = fixture(), nodes = params.nodes as JsonObject[]; + const template = nodes[0].research_observation as JsonObject; + for (const [source, target] of [["c", "d"], ["e", "f"], ["g", "h"]]) { + for (const node_id of [source, target]) { + const raw = {...template, explore_node_id: node_id, + progress: {...template.progress as JsonObject, work_item_id: `todo_${node_id}`, evidence_ids: [`ev-${node_id}`]}, + closure_basis: {...template.closure_basis as JsonObject, evidence_ids: [`ev-${node_id}`]}, + composition_candidates: node_id === source ? [{target_node_id: target, basis: "explicit", + interaction_kind: "state_interference", evidence_ids: [`ev-${source}`, `ev-${target}`]}] : []}; + nodes.push({node_id, node_kind: "hypothesis", status: "resolved", + research_observation: normalizeResearchObservation({observation: raw})}); + } + } + const all = researchCompositionGaps(params), compact = projectResearchFrontier(params); + assert.equal(all.length, 4); + assert.equal((compact.gaps as JsonObject[]).length, 3); + const omitted = all[3]; + params.edges = (omitted.input_node_ids as string[]).map(to_node => ({from_node: "joint", to_node, edge_type: "depends_on"})); + const raw = params.observation as JsonObject; + params.observation = {...raw, input_observations: omitted.input_observations, + execution_lineage: {...raw.execution_lineage as JsonObject, gap_id: omitted.gap_id}}; + assert.doesNotThrow(() => validateResearchExecution(params)); +}); diff --git a/tests/control_plane_ts/quota_settlement_readback.test.ts b/tests/control_plane_ts/quota_settlement_readback.test.ts index fd5304638b..3ddd6d856c 100644 --- a/tests/control_plane_ts/quota_settlement_readback.test.ts +++ b/tests/control_plane_ts/quota_settlement_readback.test.ts @@ -58,6 +58,78 @@ test("semantic replan guard distinguishes legacy, none, and exact selection", () ); }); +test("capability guard survives scalar receipt transport and rejects missing scope facts", () => { + const details = {semantic_replan_obligation_id: "replan-0000000000000001", + semantic_replan_capability_id: "explore", semantic_replan_gap_id: "research-composition-0123456789abcdef", + semantic_replan_frontier_revision: "research-composition-v0:0123456789abcdef"}; + assert.deepEqual(projectSemanticReplanGuard(details).selected_capability_guard, { + schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore", + gap_id: details.semantic_replan_gap_id, frontier_revision: details.semantic_replan_frontier_revision, + }); + assert.throws(() => projectSemanticReplanGuard({...details, semantic_replan_obligation_id: ""}), /selected obligation/); + assert.throws(() => projectSemanticReplanGuard({...details, semantic_replan_gap_id: ""}), /capability guard is malformed/); + assert.throws(() => projectSemanticReplanGuard({...details, semantic_replan_frontier_revision: undefined}), /capability guard is malformed/); +}); + +test("capability retirement verifies its exact receipt and never erases a committed debit", async () => { + const obligation = "replan-0000000000000001"; + const guard = {schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore", + gap_id: "research-composition-0123456789abcdef", frontier_revision: "research-composition-v0:0123456789abcdef"}; + const bound = settlementIdentity({goal_id: goalId, agent_id: agentId, todo_id: null, + replan_obligation_id: obligation, turn_instance_id: turnId}); + const progress = {schema_version: "typed_progress_observation_v0", work_item_id: obligation, + result_class: "blocked", blocker_id: "capability-invalidated-fixture", + evidence_ids: ["capability-evidence-v0:fedcba9876543210"], fingerprint: "retirement-progress"}; + const retirement = {schema_version: "capability_obligation_retirement_v0", disposition: "invalidated", + reason_code: "source_ineligible", capability_id: "explore", obligation_id: obligation, + original_guard: guard, current_revision: progress.evidence_ids[0], blocking_todo_count: 0, + blocking_todo_ids: [], progress_observation: progress, progress_fingerprint: progress.fingerprint}; + const ack = {recorded: true, source: "refresh_state_semantic_delta", semantic_delta: { + schema_version: "replan_semantic_delta_v0", accepted: true, obligation_id: obligation, + outcomes: ["capability_duty_retired"], satisfying_outcomes: ["capability_duty_retired"], capability_guard: guard, retirement}}; + const root = await fixture({writeback: true}); + try { + const index = join(root, "goals", goalId, "runs", "index.jsonl"); + const log = join(root, "goals", goalId, "rollout-event-log.jsonl"); + const events = (await readFile(log, "utf8")).trim().split("\n").map(line => JSON.parse(line)); + for (const event of events) { + event.details = {...event.details, todo_id: "", replan_obligation_id: obligation, settlement_effect_id: bound.effect_id}; + if (event.event_kind === "quota_should_run") Object.assign(event.details, { + semantic_replan_obligation_id: obligation, semantic_replan_capability_id: "explore", + semantic_replan_gap_id: guard.gap_id, semantic_replan_frontier_revision: guard.frontier_revision}); + } + await writeFile(log, events.map(event => JSON.stringify(event)).join("\n") + "\n"); + const base = {classification: "state_refreshed", goal_id: goalId, agent_id: agentId, + turn_instance_id: turnId, replan_obligation_id: obligation, settlement_identity: bound, + delivery_outcome: "outcome_gap", progress_observation: progress, autonomous_replan_ack: ack}; + const read = () => readQuotaSettlement(request(root, {todo_id: null, replan_obligation_id: obligation})); + await writeFile(index, JSON.stringify(base) + "\n"); + const accepted = await read(); + assert.equal((accepted.progress as any).closeout_kind, "capability_duty_retired_no_spend"); + assert.equal(accepted.replay_phase, "settled"); + assert.equal((accepted.spend as any).payload.ok, false); + assert.equal((accepted.terminal_closeout as any).payload.ok, false); + for (const patch of [{obligation_id: "replan-0000000000000002"}, {blocking_todo_count: 1}, + {original_guard: {...guard, frontier_revision: "research-composition-v0:fedcba9876543210"}}, + {current_revision: "unrelated"}, {progress_fingerprint: "forged"}]) { + const changed = {...base, autonomous_replan_ack: {...ack, semantic_delta: {...ack.semantic_delta, + retirement: {...retirement, ...patch}}}}; + await writeFile(index, JSON.stringify(changed) + "\n"); + assert.equal((await read()).replay_phase, "settlement_pending"); + } + await writeFile(index, JSON.stringify(base) + "\n" + JSON.stringify({classification: "quota_slot_spent", + goal_id: goalId, agent_id: agentId, turn_instance_id: turnId, replan_obligation_id: obligation, + settlement_identity: bound}) + "\n"); + await appendFile(log, JSON.stringify({schema_version: "loopx_rollout_event_v0", event_id: "spent-before-retirement", + event_kind: "quota_spend", goal_id: goalId, agent_id: agentId, run_id: turnId, + details: {settlement_effect_id: bound.effect_id}}) + "\n"); + const spent = await read(); + assert.equal((spent.progress as any).state, "settled"); + assert.equal((spent.progress as any).closeout_kind, undefined); + assert.equal((spent.spend as any).payload.ok, true); + } finally {await rm(root, {recursive: true, force: true});} +}); + async function fixture(options: { guard?: boolean; /** diff --git a/tests/control_plane_ts/replan_semantics.test.ts b/tests/control_plane_ts/replan_semantics.test.ts index f53f85890f..590f7076f2 100644 --- a/tests/control_plane_ts/replan_semantics.test.ts +++ b/tests/control_plane_ts/replan_semantics.test.ts @@ -1,6 +1,18 @@ import assert from "node:assert/strict"; import test from "node:test"; import { projectReplanSemantics, requiredSemanticOutcomes } from "../../loopx/control_plane/work_items/replan_semantics.ts"; + +test("capability evidence is not a default exit or a generic progress claim", () => { + for (const outcome of ["capability_evidence_observed", "capability_duty_retired"]) { + assert.equal(requiredSemanticOutcomes({triggers: [{kind: "no_progress_streak"}]}).includes("capability_evidence_observed"), false); + assert.throws(() => requiredSemanticOutcomes({satisfying_semantic_outcomes: ["capability_evidence_observed"]}), /bound owning capability/); + const obligation = {satisfying_semantic_outcomes: [outcome], + capability_guard: {schema_version: "semantic_replan_capability_guard_v0", capability_id: "explore"}}; + assert.deepEqual(requiredSemanticOutcomes(obligation), [outcome]); + assert.throws(() => projectReplanSemantics({operation: "qualify", obligation, + observation_delta: {delta_kinds: [outcome]}}), /owning capability, not generic progress/); + } +}); import { visionAuthoringContract } from "../../loopx/control_plane/goals/vision_checkpoint.ts"; import type { JsonObject } from "../../loopx/control_plane/effect_program.ts"; diff --git a/tests/test_chat_goal_configuration_api.py b/tests/test_chat_goal_configuration_api.py index a14964e1a4..158c7d49e9 100644 --- a/tests/test_chat_goal_configuration_api.py +++ b/tests/test_chat_goal_configuration_api.py @@ -1,6 +1,7 @@ from __future__ import annotations from pathlib import Path +import json from types import SimpleNamespace from typing import Any @@ -54,6 +55,18 @@ def _catalog_payload(*, explore_enabled: bool = False) -> dict[str, Any]: } +def test_explore_composition_options_keep_typed_scope_and_reject_coercion() -> None: + options = _goal_capability_options("explore_harness", { + "enabled": True, "composition_mode": "explicit_only", "composition_scope_id": "joint-scope", + }) + assert options["explore_composition_mode"] == "explicit_only" + assert options["explore_composition_scope_id"] == "joint-scope" + assert _goal_capability_options("explore_harness", {"enabled": True, "composition_mode": ""})["explore_composition_mode"] == "disabled" + for field, value in [("composition_mode", True), ("composition_scope_id", 0), ("composition_scope_id", [])]: + with pytest.raises(TypeError, match="must be strings"): + _goal_capability_options("explore_harness", {"enabled": True, field: value}) + + def _periodic_catalog_payload(*, override_present: bool) -> dict[str, Any]: feature: dict[str, Any] = { "feature_id": "periodic_report", @@ -421,6 +434,57 @@ def test_goal_configuration_apply_rejects_stale_preview() -> None: assert handler.responses[0]["error_code"] == "goal_configuration_preview_stale" +def test_research_policy_api_applies_and_reads_real_source_and_shared_registry(tmp_path: Path) -> None: + project, runtime, shared = tmp_path / "project", tmp_path / "runtime", tmp_path / "shared" + registry = project / ".loopx" / "registry.json" + registry.parent.mkdir(parents=True) + registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{ + "id": "goal-example", "repo": str(project), "status": "active", + "spawn_policy": {"explore_harness": {"enabled": False}}, + }]})) + + class RealHandler(_MutationHandler): + _goal_configuration_reader = GoalConfigurationRequestMixin._goal_configuration_reader + _goal_configuration_writer = GoalConfigurationRequestMixin._goal_configuration_writer + + def __init__(self, body): + super().__init__(body) + self.server = SimpleNamespace(registry_path=registry, runtime_root=runtime, + runtime_root_override=str(shared)) + + def _goal_configuration_machine_namespaces(self): + return [] + + for configuration in [ + {"enabled": True, "composition_mode": "explicit_only", "composition_scope_id": "joint-scope"}, + {"enabled": True, "composition_mode": "disabled", "composition_scope_id": "joint-scope"}, + ]: + body = {"goal_id": "goal-example", "capability_id": "explore_harness", "configuration": configuration} + before = registry.read_bytes() + preview = RealHandler(body) + preview._goal_configuration_update(execute=False) + plan = preview.responses[0] + assert plan["status_code"] == 201, plan + assert registry.read_bytes() == before + applied = RealHandler({**body, "expected_plan_revision": plan["plan_revision"]}) + applied.path = CHAT_GOAL_CONFIGURATION_APPLY_PATH + applied._goal_configuration_update(execute=True) + receipt = applied.responses[0] + assert receipt["status_code"] == 200, receipt + assert receipt["readback_verified"] is True + assert {key: receipt["goal_configuration"][key] for key in configuration} == configuration + source = json.loads(registry.read_text())["goals"][0]["spawn_policy"]["explore_harness"] + mirror = json.loads((shared / "registry.global.json").read_text())["goals"][0]["spawn_policy"]["explore_harness"] + assert {key: source[key] for key in configuration} == configuration + assert mirror == source + stale = RealHandler({**body, "expected_plan_revision": plan["plan_revision"]}) + stale.path = CHAT_GOAL_CONFIGURATION_APPLY_PATH + after = registry.read_bytes(), (shared / "registry.global.json").read_bytes() + stale._goal_configuration_update(execute=True) + assert stale.responses[0]["status_code"] == 409 + assert (registry.read_bytes(), (shared / "registry.global.json").read_bytes()) == after + + def test_goal_configuration_clear_override_is_revision_locked() -> None: body = { "goal_id": "goal-example", diff --git a/tsconfig.control-plane.json b/tsconfig.control-plane.json index 4ec2310ab9..e2d3b43664 100644 --- a/tsconfig.control-plane.json +++ b/tsconfig.control-plane.json @@ -15,6 +15,8 @@ "include": [ "loopx/control_plane/capabilities/explore_research.ts", "tests/control_plane_ts/explore_research.test.ts", + "loopx/control_plane/capabilities/explore_research_execution.ts", + "tests/control_plane_ts/explore_research_execution.test.ts", "loopx/control_plane/runtime/usage_statistics*.ts", "tests/control_plane_ts/usage_statistics.test.ts", "tests/control_plane_ts/usage_statistics_goals.test.ts", From 21a2d9d3fc5403c653f2aa108f578f3c5e3d1831 Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Mon, 28 Sep 2026 21:36:40 -0700 Subject: [PATCH 03/12] docs(explore): document scoped lineage and no-spend retirement Signed-off-by: Lihua <1017343802@qq.com> --- .../research-exploration-control-plane-v0.md | 40 +++- ...arch-exploration-control-plane-v0.zh-CN.md | 33 ++- docs/reference/handoff-mode.md | 16 ++ .../protocols/research-observation-v0.md | 195 +++++++++++++++++- .../research-observation-v0.zh-CN.md | 165 ++++++++++++++- 5 files changed, 425 insertions(+), 24 deletions(-) diff --git a/docs/architecture/rfcs/research-exploration-control-plane-v0.md b/docs/architecture/rfcs/research-exploration-control-plane-v0.md index 2c5bfa54ac..dc8590a138 100644 --- a/docs/architecture/rfcs/research-exploration-control-plane-v0.md +++ b/docs/architecture/rfcs/research-exploration-control-plane-v0.md @@ -150,10 +150,8 @@ milestone status. | Gap | Consequence | |---|---| -| No shared research write-time gate | The cold evidence codec cannot discharge or enforce a live composition obligation. | -| No exact obligation/Todo/result lineage | A research receipt is not proof of an authorized Todo transition or accepted Goal closure. | -| Cold shadow not adopted by hot status/frontier | The existing #3173 projection remains behavior-compatible; canonical research obligations still need M3 integration. | -| No dismissal or deferral contract | Evidence invalidation is visible, but typed candidate retirement and resumption remain unimplemented. | +| Integrated M3 qualification remains | Typed dismissal, blocker waits and exact invalidated-duty retirement are implemented locally; final premerge and maintainer-reviewed integration remain distinct from implementation and live research qualification. | +| Execution attribution is not effect authority | The live replan gate joins exact Todo/experiment/input facts; its receipt is not task-lease/effect authorization or accepted Goal closure. | | Live qualification incomplete | Deterministic and real CLI/file-log tests establish state semantics, not model selection quality or scientific truth; no live Lark sync is qualified by projection tests. | | No promotion evidence for inferred combinations | Shared constraints are not known to be precise enough to trigger obligations. | @@ -232,8 +230,9 @@ execution, and result into one ambiguous relation. This RFC rejects that shape. The research envelope and closure basis have an active CLI caller and the [versioned evidence protocol](../../reference/protocols/research-observation-v0.md). -The action signature, shared write gate and model selection below remain design -targets. The cold shadow does not promote them into current behavior. +The opt-in M3 development path implements exact execution lineage and shared +write gates and bounded retirement transitions. Model selection below +remain design targets; the cold shadow does not activate enforcement. ### 7.1 Compose; do not mutate v0 silently @@ -804,7 +803,7 @@ control-plane failures. | M0 | RFC, current-state inventory, and explicit ownership decision | Maintainer review; no runtime behavior | Accepted design | | M1 | Characterization fixtures plus typed research observation and closure contract in Explore | Deterministic normalization, privacy, compatibility, and negative tests | Implemented evidence/CLI slice; live research qualification remains separate | | M2 | Explicit-only composition candidate, canonical gap projection, and read-only status shadow | No pairwise inference; bounded packet; projection parity | Partial: #3173 legacy quota/successor; canonical binary cold shadow in CLI/Lark projection; hot status adoption and live Lark qualification remain | -| M3 | Goal-frontier obligation, exact Todo/experiment lineage, and shared write-time gate | State/replay matrix and premerge canary pass | Not started | +| M3 | Goal-frontier obligation, exact Todo/experiment lineage, and shared write-time gate | State/replay matrix and premerge canary pass | Local state/replay and standard premerge qualified; maintainer-reviewed integration pending. Scoped policy/editor, shared gates, archive lineage, consumers, dismissal, waits and exact no-spend duty retirement are implemented | | M4 | Bounded multi-candidate cards, `composition_selection_v0`, real model-tool behavior qualification, and repeated live shadow | Model autonomously selects a legal semantic action from the delivered candidate set; selection quality is no worse than the declared fallback; compact receipts only | Not started | | M5 | Shared-constraint candidate ranking in shadow mode | Precision and cost evidence; no automatic trigger | Not started | | M6 | Optional inferred trigger | Explicit maintainer decision and measured promotion thresholds | Deferred | @@ -835,6 +834,33 @@ M3 is the first behavior-changing slice. It should be a separate PR so the obligation and write gate can be reviewed and reverted independently from the evidence schema. +The M3 development boundary joins current same-agent Todo, experiment and input +facts in the typed Explore owner. Goal policy is explicitly scoped and disabled +by default; the existing capability editor and CLI share its configuration +owner. Quota and refresh reuse one live frontier, with the original duty pinned +through scalar rollout fields and both receipt adapters. Real CLI tests reject +unrelated/deferred successors and invalidated writeback, then settle the original +Turn through its exact successor. File/SQLite tests distinguish canonical Todos +from stale display rows; packaged UI checks exercise policy preview, activation, +disable, readback and narrow screens. Native actor/lease and CAS admission remain in force; a fixed IO host +locks the graph while the typed owner qualifies and persists completion evidence. +Real File/SQLite and legacy tests cover missing evidence, direct IPC self-approval, +terminal-verb bypass, retained archive lineage and immutable completion replay. +Agent-scoped status, Explore and existing Lark Summary fields use the same live +facts. Typed candidate dismissal permits scoped terminal retirement; a fresh +canonical blocker and common Todo resume condition defer the gap without closing +it. Real CLI validates exact observed/dismissed/blocked progress source through +the original Turn, with independent File/SQLite lease-safe wait/resume evidence. +Invalidated input/scope/activation produces source-qualified retirement for the +original duty. Common readback retains its guard and historical debit, closes +that Turn without spend, and keeps the current frontier and runnable/paused work +visible. IO assembly stays in existing CLI/refresh composition roots; the shared +gate consumes supplied capability facts and the typed owner remains singular. +The local state/replay matrix and 19 standard risk-selected premerge checks pass. +Two existing scheduler ACK tests fail identically on the unchanged base under the +same local runtime; this is retained as a baseline limitation, not called green. +Maintainer-reviewed integration and independent live qualification remain. + ## 17. Rejected Alternatives ### 17.1 Automatically pair nodes with a shared closure stage diff --git a/docs/architecture/rfcs/research-exploration-control-plane-v0.zh-CN.md b/docs/architecture/rfcs/research-exploration-control-plane-v0.zh-CN.md index a497f608fa..0029a5bb07 100644 --- a/docs/architecture/rfcs/research-exploration-control-plane-v0.zh-CN.md +++ b/docs/architecture/rfcs/research-exploration-control-plane-v0.zh-CN.md @@ -131,10 +131,8 @@ hypothesis”或“新 probe family”。它无法持久表达:A 和 B 都已 | 缺口 | 后果 | |---|---| -| 没有 shared research write-time gate | Cold evidence codec 不能解除或强制 live composition obligation。 | -| 没有精确 obligation/Todo/result lineage | Research receipt 不证明已授权 Todo transition 或已接受 Goal closure。 | -| Cold shadow 未接入 hot status/frontier | 既有 #3173 投影保持行为兼容;canonical research obligation 仍需 M3 集成。 | -| 没有 dismissal/deferral contract | Evidence invalidation 可见,但类型化 candidate retirement/resumption 尚未实现。 | +| M3 集成资格仍未完成 | Typed dismissal、blocker wait 和精确 invalidated-duty retirement 已在本地实现;最终 premerge 与 maintainer review 集成仍独立于实现和真实研究 qualification。 | +| 执行 attribution 不是 effect 权限 | Live replan 门禁连接精确 Todo/experiment/input 事实;receipt 不证明 task-lease/effect 授权或已接受 Goal closure。 | | Live qualification 不完整 | Deterministic 与真实 CLI/file-log 测试证明状态语义,不证明 model selection 质量或科学结论;projection 测试不构成 live Lark sync 资格。 | | inferred combination 没有 promotion evidence | 共享 constraint 的精度还不足以直接触发 obligation。 | @@ -209,8 +207,8 @@ A、B 之间的 `joint_probe` 直连边会把 candidate、execution 和 result Research envelope 与 closure basis 已有真实 CLI caller 和 [版本化证据协议](../../reference/protocols/research-observation-v0.zh-CN.md)。 -下文 action signature、shared write gate 和 model selection 仍为设计目标; -cold shadow 不会将其 promotion 为当前行为。 +Opt-in M3 开发路径已实现精确 execution lineage 和共享 write gate。下文 model +selection 仍为设计目标;有界 retirement 已实现,cold shadow 不会激活门禁。 ### 7.1 组合,而不是静默修改 v0 @@ -742,7 +740,7 @@ rule,以及 model variance 与 control-plane failure 的分离。 | M0 | RFC、current-state inventory 与显式 ownership decision | Maintainer review;无 runtime behavior | 已接受的设计 | | M1 | Characterization fixture,以及 Explore 中的 typed research observation 与 closure contract | Deterministic normalization、privacy、compatibility 与 negative test | Evidence/CLI 切片已实现;真实研究 qualification 独立保留 | | M2 | Explicit-only composition candidate、canonical gap projection 与 read-only status shadow | 不做 pairwise inference;packet 有界;projection parity | 部分实现:#3173 legacy quota/successor;CLI/Lark projection 的 canonical binary cold shadow;hot status adoption 与 live Lark qualification 仍未完成 | -| M3 | Goal-frontier obligation、精确 Todo/experiment lineage 与共享 write-time gate | State/replay matrix 与 premerge canary 通过 | 未开始 | +| M3 | Goal-frontier obligation、精确 Todo/experiment lineage 与共享 write-time gate | State/replay matrix 与 premerge canary 通过 | 本地 state/replay 与 standard premerge 已验证,maintainer review 集成待完成。Scoped policy/editor、共享门禁、archive lineage、消费者、dismissal、wait 与精确无支出 duty retirement 已实现 | | M4 | 有界 multi-candidate card、`composition_selection_v0`、真实 model-tool behavior qualification 与重复 live shadow | 模型从交付 candidate set 中自主选择合法 semantic action;选择质量不劣于 declared fallback;只保留 compact receipt | 未开始 | | M5 | Shared-constraint candidate 在 shadow mode 中排序 | 有 precision/cost evidence;不自动触发 | 未开始 | | M6 | 可选 inferred trigger | 显式 maintainer decision 与量化 promotion threshold | 延后 | @@ -771,6 +769,27 @@ Projection 测试不证明 live remote effect 或模型自主研究行为。交 M3 是第一个 behavior-changing slice。它应单独成 PR,使 obligation 与 write gate 能够独立于 evidence schema 评审和回滚。 +M3 开发边界在 typed Explore owner 中连接当前同一 Agent 的 Todo、experiment +和 input 事实。Goal policy 显式限定范围且默认关闭;既有能力编辑器与 CLI 共享 +配置 owner。Quota 与 refresh 复用同一 live frontier,原 duty 通过 rollout 标量 +字段和两端 receipt adapter 固定。真实 CLI 测试拒绝无关/延期 successor 与失效 +writeback,再通过精确 successor 结算原 Turn。File/SQLite 测试区分 canonical +Todo 与陈旧显示行;打包 UI 验证策略预览、启用、关闭、读回和窄屏。Native actor/lease 与 CAS admission +仍强制执行;固定 IO host 锁定图,typed owner 校验并持久化 completion evidence。 +真实 File/SQLite 与 legacy 测试覆盖缺少证据、direct IPC 自报 approval、terminal +verb 绕过、保留 archive lineage 和不可变 completion 回放。Agent 范围的 status、 +Explore 与既有 Lark Summary 字段使用同一 live fact。Typed candidate dismissal +允许 scoped terminal retirement;新 canonical blocker 与通用 Todo resume condition +使 gap 暂缓,但不关闭。真实 CLI 通过原 Turn 校验 observed/dismissed/blocked 的 +精确 progress source;File/SQLite 独立验证 lease-safe wait/resume。 +输入/scope/activation 失效时,为原 duty 生成 source-qualified retirement。 +通用 readback 保留原 guard 与历史 debit,无支出关闭该 Turn,同时保持当前 +frontier 与 runnable/paused work 可见。IO 装配保留在既有 CLI/refresh composition +root;共享门禁消费传入的 capability fact,typed owner 始终只有一个。 +本地 state/replay matrix 与 19 项 standard 风险验证通过。两个既有 scheduler ACK +测试在相同本地 runtime、未修改 base 上也以相同方式失败;保留为 baseline 限制, +不称为 green。Maintainer review 集成与独立 live qualification 仍未完成。 + ## 17. 被拒绝的替代方案 ### 17.1 自动组合拥有共享 closure stage 的 node diff --git a/docs/reference/handoff-mode.md b/docs/reference/handoff-mode.md index 161df7f5e3..f0c7519c7c 100644 --- a/docs/reference/handoff-mode.md +++ b/docs/reference/handoff-mode.md @@ -207,6 +207,22 @@ permission. The original rejection code and all fences remain unchanged: | `soft_claim`, non-open Todo, acceptance hold or conflicting write scopes | Resolve the reported acquisition blocker. No acquire action is offered. | | Edit changes retained leased work requirements or status | Use the owning lifecycle transition; acquiring another lease cannot authorize the metadata edit. | +The existing `open -> blocked -> open` lifecycle also accepts a typed +prerequisite wait: use `todo update --status blocked --resume-when +todo_done: --reason ''` after the active holder +releases its execution lease. This explicit wait form is newly supported for +native hard-lease Todos; the prior clear-wait pause form is unchanged. A live +lease or bundled execution proof still rejects the transition. Resume with +`--status open --clear-resume-when --reason ''`; neither transition +grants execution authority, and the next execution requires a fresh lease. + +既有 `open -> blocked -> open` lifecycle 也支持 typed 前置任务等待:active +holder 先释放执行 lease,再用 `todo update --status blocked --resume-when +todo_done: --reason ''` 暂停。此显式 wait 形式 +新增支持 native hard-lease Todo;原 clear-wait pause 形式保持不变。有效 lease +或附带执行 proof 仍被拒绝。用 `--status open --clear-resume-when --reason +''` 恢复;两次转换都不授予执行权限,下一次执行仍需新 lease。 + The recovery descriptor uses the standalone `loopx task-lease acquire` command. Combined `todo claim --task-lease-idempotency-key` is restricted to `hard_lease` and is not the recovery route for `legacy`. Acquire uses `--owner`, a **fresh** diff --git a/docs/reference/protocols/research-observation-v0.md b/docs/reference/protocols/research-observation-v0.md index d307ff4bbd..f7b01dd247 100644 --- a/docs/reference/protocols/research-observation-v0.md +++ b/docs/reference/protocols/research-observation-v0.md @@ -3,7 +3,9 @@ Explore owns this optional evidence contract. Its first consumer is `loopx explore observe`; `loopx explore summary` and the existing Lark Explore node summary render the same derived facts. This is the M1 evidence substrate -and M2 **read-only shadow**, not M3 obligation/writeback enforcement. +and M2 **read-only shadow**. The M3 development boundary adds optional execution +attribution, an explicit-only live replan gate, and legacy/native File/SQLite +Todo closeout checks. Retirement/resumption and full status/Explore presentation remain open. ## Record and read back @@ -83,7 +85,7 @@ optional `research_observation` on an append-only node revision. Existing events without that field preserve their projection shape and require no research runtime operation. Older readers that validate node fields strictly must upgrade before reading a log containing the optional envelope. Existing -`#3173` composition/quota behavior is unchanged. No research policy, scheduler, +`#3173` composition/quota behavior is unchanged without the new policy. No research policy, scheduler, claim, lease, quota, generic settlement or Goal acceptance is enabled by this command. @@ -112,9 +114,192 @@ the node writer and batch writer enforce attribution for typed envelopes. The cold projection returns total counts and at most three cards, with explicit `projected_count` and `omitted_count`. It makes no ranking-quality claim and does not add a full candidate list to quota packets. Inspect the canonical -node observations to investigate omitted candidates; no obligations are lost -because this slice creates none. Dismissal, deferral, observation-to-Todo -lineage, shared write gates and hot status adoption remain M3 work. +node observations to investigate omitted candidates. Cold cards do not select +obligations. The live policy below uses the full internal candidate set; +compaction cannot erase an enforceable gap. + +## Execution attribution + +An experiment may additionally record `execution_lineage`: + +```json +{ + "schema_version": "research_execution_lineage_v0", + "goal_id": "research-demo", + "gap_id": "research-composition-0123456789abcdef", + "replan_obligation_id": "replan-0123456789abcdef", + "successor_todo_id": "todo_joint", + "agent_id": "research-agent" +} +``` + +Use the actual gap, obligation and Todo identities supplied by the current +work contract; the example tokens do not establish an obligation. Include this +object in the experiment envelope and issue `explore observe --agent-id` with +the same actor. `progress.work_item_id` must equal `successor_todo_id`. +The source registry must contain that Goal. The writer reads the exact Todo +through the existing Todo reader, including promoted canonical authority when +configured; it does not accept a caller-supplied Todo snapshot. + +A new write requires a same-agent, currently runnable `advancement_task` with +`action_kind=joint_probe`, the exact `replan_obligation_id`, and exactly this +experiment in `explore_result_node_refs`. A declared `target_key` must also +equal the experiment id. Deferred, blocked, rejected by the existing acceptance +guard, archived, executor-excluded, other-agent or unrelated work cannot supply +execution attribution. The experiment must belong to this pending binary gap +and match its current input observations. An unchanged/read/ACK observation or +an execution result without evidence is rejected. + +The fingerprint includes this object, research schema, experiment identity, +input fingerprints and generic progress/evidence. Node and batch writers +reject new execution-lineage writes; use `explore observe` for its task read. +Exact historical replay still writes nothing, even after task or input changes. +It does not refresh current task authority. Presentation compaction does not +limit the internal candidate set used for attribution. + +This is a task snapshot attribution check under the Explore log lock, not an +atomic task lease/effect authorization or a Todo/Goal completion receipt. +Any subsequent shared settlement must independently validate its current +authority and research evidence. Envelopes without `execution_lineage` retain +their diagnostic behavior; they cannot be promoted into execution lineage by +reading or replaying them. The live replan gate below adopts these facts; +the same typed rule also guards current legacy and native File/SQLite Todo closeout. + +Completed Todo records retain evidence lineage after archival or claim clearing. +The projection reads retained canonical history rather than a compact active +Todo list. An archived open/deferred row cannot schedule or attribute a new +observation; missing or mismatched history and invalidated inputs still reject +current coverage. Historical evidence grants no current claim, lease or execution +authority. + +## Explicit-only live replan gate (M3 development) + +Preview and apply the policy through the existing Goal configuration owner: + +```sh +loopx configure-goal --goal-id research-demo --explore-harness-enabled \ + --explore-composition-mode explicit_only --explore-composition-scope-id joint-scope +loopx configure-goal --goal-id research-demo --explore-harness-enabled \ + --explore-composition-mode explicit_only --explore-composition-scope-id joint-scope --execute +loopx quota should-run --goal-id research-demo --agent-id research-agent --turn-instance-id research-turn +``` + +The Goal capability editor exposes the same activation, policy and scope fields +through its existing revision-checked preview/apply API. Both harness activation +and `composition_mode=explicit_only` are required. The opaque +`composition_scope_id` identifies experiment coverage; it does not replace the +Goal vision or grant claim, lease, effect, provider or external-execution rights. + +Quota and `refresh-state` read the same live Explore/Todo frontier. A compact +capability guard pins the selected gap and input/policy revision to the original +Turn. The existing common replan owner computes its obligation id. User/handoff +gates, runnable work, vision acceptance and ordinary succession/review duties +retain priority before this capability gap enters monitor fallback. + +Only an exact same-agent runnable `joint_probe` successor with the current +obligation and one current binary experiment suppresses duplicate planning. +`todo add --replan-obligation-id` rejects unrelated or deferred successors. +Scheduling does not observe the gap. A terminal experiment result must include +execution lineage, current input fingerprints and the configured coverage scope +before it becomes live observed evidence. Its complete generic progress and +research fingerprint bind writeback; omitted semantics, unrelated progress, +reads/ACKs, stale inputs and replay cannot discharge the selected duty. + +The supported replan exits are a new runnable experiment successor, its typed +observed result, an evidence-backed candidate dismissal or a fresh exact blocker +with a typed resume condition. A negative scientific conclusion is valid evidence. +Generic blocked/terminal claims do not certify research closure. Current legacy +and native File/SQLite closeout require the task's exact terminal experiment or +dismissal evidence. An invalidated admitted duty uses the lifecycle retirement +below; a Todo becoming done never certifies Goal closure. + +An observation may include `composition_resolution` with schema +`research_composition_resolution_v0`. It requires execution lineage and nonempty +`evidence_ids` attributable to the same progress observation: + +- `disposition=dismissed` requires coverage-backed `no_followup`, a matching + `no_followup` closure basis, an experiment node in `dead_end` and typed `basis` + (`duplicate`, `invalid`, `unsafe` or `outside_scope`). It counts as `dismissed`, + never an observed experiment. The exact evidence permits normal Todo completion + or supersession without pretending the experiment ran. +- `disposition=deferred` requires `blocked` progress and its canonical + `blocker_id`; it cannot assert complete terminal coverage. The linked experiment + is blocked, its current Todo is blocked/deferred with + `resume_when=todo_done:`, and the same-agent active blocker Todo + names that experiment Todo in `unblocks_todo_id`. The common Todo resume owner + must prove the prerequisite is still pending. Missing, unrelated, archived or + already-resolved blockers cannot suppress the gap. Reusing a previously claimed + blocker cannot provide fresh writeback progress. + +These are caller-declared evidence dispositions, not independent scientific +verification or execution permission. Resolving the prerequisite exposes the +pending gap; reopening the task and experiment makes it scheduled, not observed. +Native hard-lease callers must release their active lease before the existing +blocked lifecycle accepts a typed wait. Resume clears the wait and still requires +a fresh execution lease. Arbitrary deferred successors remain rejected. + +When refreshing a replan-bound Turn from a canonical experiment observation, +use `--progress-work-item-id` for the observation's exact source Todo and repeat +its complete generic progress fields. This selector does not rebind settlement +away from the original obligation. The shared gate still verifies the actual +Todo, current inputs, scope, evidence and fingerprint; changed or incomplete +observations are rejected. A new concrete blocker is an explicit replan progress +exit; it does not use the Todo-bound `outcome_gap` no-spend closeout contract. + +The native transaction evaluates capability eligibility against its actual +provider Todo after actor/lease admission and before caller validation effects. +A fixed source-selected Python adapter holds the canonical Explore log lock +through the native CAS. It only reads the graph; TypeScript decides whether +the observation, task and input lineage match. The model cannot supply an +approval flag, snapshot or executable to bypass this guard. Accepted evidence +is retained in the terminal operation receipt. Replaying that receipt reports +history even after later input invalidation, without rewriting state or evidence. +Completing through a different terminal verb cannot avoid the same check; +typed candidate dismissal above is a legal retirement without experimental outcome evidence. + +The writer holds the Explore log lock while qualifying and persisting its +research writeback. The persisted guard uses bounded scalar rollout fields; +both TS and Python receipt adapters retain the same selected facts. Public +quota cards omit internal task joins, transition candidates and full result +observations, with counts for omitted display cards. + +`status --goal-id research-demo --agent-id research-agent` displays these same +live counts and bounded gaps in `project_asset.bounded_research_frontier`. +`explore summary --goal-id research-demo --agent-id research-agent` exposes +`research_execution_frontier` and annotates existing node summaries. The same +summaries feed Lark projection fields; `feishu-sync` and `feishu-card` accept +the same Agent selector. A live-policy Goal with multiple registered Agents +requires that explicit selector; a single Agent is selected automatically for +presentation only. These read paths do not start a Turn, mutate Todo state or +authorize remote writes. Inactive policy preserves the existing projections. + +Disable only the new policy while retaining the existing planner: + +```sh +loopx configure-goal --goal-id research-demo --explore-composition-mode disabled --execute +``` + +Removing the policy fields also restores the existing behavior. Changing scope, +inputs or activation cannot retroactively erase an admitted Turn's guard; stale +writeback remains rejected. Its rejection supplies an exact `retirement_contract` +when current source facts prove invalidation. Reuse its blocked progress fields +with the original `--replan-obligation-id` and `--turn-instance-id`, and set +`--delivery-outcome outcome_gap`. The capability owner revalidates the current +source and generates `capability_obligation_retirement_v0`; caller approval flags +or arbitrary blocker/evidence ids cannot provide this proof. Runnable Todos bound +to that duty prevent retirement until their owning lifecycle pauses them. +Invalid, missing or unavailable evidence sources cannot stand in for invalidation. + +The common settlement owner verifies the original guard, durable writeback, +current revision and exact progress fingerprint, then returns +`closeout_kind=capability_duty_retired_no_spend`. It closes only that admitted +Turn. No Todo or Goal is completed, and no quota slot is spent; already committed +debits remain ordinary historical debits. Old-Turn reentry and spend requests +return the same no-debit closeout without appending accounting. The next Turn +reassesses the current frontier, including remaining acceptance and paused work. +Keep the original receipt and evidence rather than deleting them to manufacture +settlement. No policy setting authorizes Goal +termination, live model qualification, benchmark launch, deployment or release. ## Disable and authority boundary diff --git a/docs/reference/protocols/research-observation-v0.zh-CN.md b/docs/reference/protocols/research-observation-v0.zh-CN.md index 37a0664236..d7e0b64d26 100644 --- a/docs/reference/protocols/research-observation-v0.zh-CN.md +++ b/docs/reference/protocols/research-observation-v0.zh-CN.md @@ -2,7 +2,9 @@ Explore 拥有这一可选证据契约。第一个写入入口是 `loopx explore observe`; `loopx explore summary` 与现有 Lark Explore 节点摘要显示同一派生事实。 -本批交付 M1 证据基础和 M2 **只读 shadow**,不实现 M3 的 obligation/writeback 门禁。 +本批交付 M1 证据基础和 M2 **只读 shadow**。M3 开发边界增加可选执行 attribution +与 explicit-only live replan 门禁,以及 legacy/native File/SQLite Todo closeout +校验。Retirement/resumption 及完整 status/Explore 展示仍未完成。 ## 录入与读回 @@ -74,7 +76,7 @@ Candidate 必须 `basis=explicit`,引用不同的已知 target,且 evidence 接受的 envelope 获得内容 fingerprint,作为 append-only 节点 revision 的可选 `research_observation` 存储。没有该字段的旧 event 保持 projection 形状,无需调用 research runtime。严格校验节点字段的旧 reader 必须升级后才能读取含新 envelope -的日志。既有 `#3173` composition/quota 行为不变;此命令不启用 research policy、 +的日志。没有新策略时,既有 `#3173` composition/quota 行为不变;此命令不启用 research policy、 scheduler、claim、lease、quota、generic settlement 或 Goal acceptance。 ## 结果、失效与回放 @@ -96,9 +98,162 @@ writer 和 batch writer 同样校验 typed envelope 的 attribution。 Cold projection 返回总数和至多三张 card,并提供 `projected_count`、`omitted_count`。 不声明排序质量,不把完整 candidate 列表加入 quota packet。遗漏项可从 canonical -node observation 检查;本批不创建 obligation,因此不会丢失义务。Dismissal、 -deferral、observation/Todo lineage、shared write gate 和 hot status adoption -仍属 M3。 +node observation 检查。Cold card 不选择 obligation;下述 live policy 使用完整 +内部候选集合,显示截断不能抹去 enforceable gap。 + +## 执行 attribution + +实验 envelope 可额外包含 `execution_lineage`: + +```json +{ + "schema_version": "research_execution_lineage_v0", + "goal_id": "research-demo", + "gap_id": "research-composition-0123456789abcdef", + "replan_obligation_id": "replan-0123456789abcdef", + "successor_todo_id": "todo_joint", + "agent_id": "research-agent" +} +``` + +使用当前工作契约提供的真实 gap、obligation 和 Todo 身份;示例 token 不产生 +义务。将该对象放入 experiment envelope,并用同一 actor 发出 +`explore observe --agent-id`。`progress.work_item_id` 必须等于 +`successor_todo_id`。源 registry 必须包含该 Goal。Writer 通过既有 Todo reader +读取精确任务,包括已配置的 promoted canonical authority,不接受调用者自报的 +Todo 快照。 + +新写入要求当前可执行、归属同一 Agent 的 `advancement_task`,其 +`action_kind=joint_probe`、`replan_obligation_id` 精确匹配,且 +`explore_result_node_refs` 仅包含本实验。声明的 `target_key` 也必须等于实验 id。 +延期、阻塞、被既有 acceptance guard 拒绝、归档、executor 被排除、其他 Agent +或无关任务均不能提供执行 attribution。实验必须对应这个 pending 二元 gap, +精确匹配当前输入 observation。Unchanged/read/ACK 或无 evidence 的结果被拒绝。 + +Fingerprint 包含该对象、research schema、实验身份、输入 fingerprint 和 generic +progress/evidence。Node 和 batch writer 拒绝新的 execution-lineage 写入,须使用 +`explore observe` 的任务读取路径。精确历史回放始终不写入,即使任务或输入已经 +改变,也不刷新当前任务权限。显示 card 的截断不限制内部 attribution 候选集合。 + +这是 Explore 日志锁内的任务快照 attribution 校验,不是原子的任务 lease/effect +授权,也不是 Todo/Goal 完成收据。后续 shared settlement 必须独立校验当前权限和 +研究证据。不带 `execution_lineage` 的 envelope 保持诊断行为,不能经读取或回放 +升级为执行 lineage。下述 live replan 门禁采用这些事实;相同 typed rule +也控制当前 legacy/native File/SQLite Todo closeout。 + +## Explicit-only live replan 门禁(M3 开发中) + +通过既有 Goal 配置 owner 预览、应用策略: + +```sh +loopx configure-goal --goal-id research-demo --explore-harness-enabled \ + --explore-composition-mode explicit_only --explore-composition-scope-id joint-scope +loopx configure-goal --goal-id research-demo --explore-harness-enabled \ + --explore-composition-mode explicit_only --explore-composition-scope-id joint-scope --execute +loopx quota should-run --goal-id research-demo --agent-id research-agent --turn-instance-id research-turn +``` + +Goal 能力编辑器通过既有 revision-checked preview/apply API 展示相同的启用、策略 +与范围字段。Harness 启用与 `composition_mode=explicit_only` 必须同时满足。 +不含私有内容的 `composition_scope_id` 指定实验覆盖范围;不替代 Goal vision, +不授予 claim、lease、effect、provider 或外部执行权限。 + +Quota 与 `refresh-state` 读取同一 live Explore/Todo frontier。Compact capability +guard 将所选 gap 与 input/policy revision 固定在原 Turn;既有 common replan +owner 计算 obligation id。User/handoff gate、可执行工作、vision acceptance 和 +普通 succession/review duty 保持更高优先级,之后 capability gap 才进入 monitor fallback。 + +仅当前同一 Agent、精确 obligation、单个当前二元实验的可执行 `joint_probe` +successor 能抑制重复规划。`todo add --replan-obligation-id` 拒绝无关或延期的 +successor。安排工作不等于 gap 已 observed。Terminal 实验结果须包含 execution +lineage、当前 input fingerprint 和配置的 coverage scope,才能成为 live observed +evidence。完整 generic progress 与 research fingerprint 绑定 writeback;丢失语义、 +无关 progress、read/ACK、过期输入及回放不能解除所选义务。 + +此边界支持可执行实验 successor、其 typed observed result、有证据的 candidate +dismissal 或带 typed resume condition 的新精确 blocker。科学结论为负也是有效 +证据。Generic blocked/terminal claim 不证明研究 closure。当前 legacy/native +File/SQLite closeout 需要该任务的精确 terminal 实验或 dismissal 证据;失效的已 +准入 duty 使用下述 lifecycle retirement。Todo done 始终不证明 Goal closure。 + +Observation 可含 schema 为 `research_composition_resolution_v0` 的 +`composition_resolution`。它需要 execution lineage 和非空 `evidence_ids`, +且证据必须归属于同一 progress observation: + +- `disposition=dismissed` 要求 coverage-backed `no_followup`、匹配的 + `no_followup` closure basis、`dead_end` 实验节点与 typed `basis` + (`duplicate`、`invalid`、`unsafe` 或 `outside_scope`)。它计为 `dismissed`, + 不计为 observed experiment。精确证据允许正常 Todo completion 或 supersession, + 不暗示实验已运行。 +- `disposition=deferred` 要求 `blocked` progress 与 canonical `blocker_id`, + 不得断言完整 terminal coverage。实验节点为 blocked;当前实验 Todo 为 + blocked/deferred,声明 `resume_when=todo_done:`;同一 Agent 的 + 活动 blocker Todo 通过 `unblocks_todo_id` 指向实验 Todo。通用 Todo resume + owner 必须证明该前置任务仍 pending。缺失、无关、归档或已解决的 blocker + 不能压制 gap;已在 writeback 中使用过的 blocker 不能再作为新进展。 + +这些是调用者声明的证据 disposition,不是独立科学验证或执行许可。前置任务 +解决后 gap 重新 pending;重新打开任务和实验仅使其 scheduled,不是 observed。 +Native hard-lease 调用者必须先释放 active lease,既有 blocked lifecycle 才接受 +typed wait。恢复清除 wait 后仍需新的执行 lease;任意 deferred successor 仍被拒绝。 + +从 canonical experiment observation 写回 replan-bound Turn 时,用 +`--progress-work-item-id` 指定其精确源 Todo,并重述完整 generic progress 字段。 +它不把 settlement 从原 obligation 重新绑定到其他工作。共享门禁仍校验真实 +Todo、当前输入、scope、evidence 与 fingerprint;修改或不完整 observation 被拒绝。 +新 concrete blocker 是显式 replan progress exit,不使用 Todo-bound +`outcome_gap` 无支出 closeout 契约。 + +Native 事务在 actor/lease admission 后、caller validation effect 前,使用实际 +provider Todo 校验能力资格。固定的 source-selected Python adapter 持有 canonical +Explore 日志锁直到 native CAS 返回;它只读取图,observation/task/input lineage +是否合法由 TypeScript 判断。模型不能自报 approval flag、snapshot 或 executable +绕过门禁。接受的证据保留在 terminal operation receipt 中。后续输入失效后,精确 +receipt 回放仍只报告历史,不重新写状态或证据。换用其他 terminal verb 也不能避开 +相同校验;上述 typed candidate dismissal 是没有实验结果证据时的合法 retirement。 + +Writer 在校验、持久化 research writeback 时持有 Explore 日志锁。持久化 guard +使用有界 rollout 标量字段;TS 与 Python receipt adapter 保留相同所选事实。 +Public quota card 不携带内部 task join、transition candidate 或完整 result +observation,并提供遗漏 card 的数量。 + +已完成 Todo 在归档或清除 claim 后保留证据 lineage。投影读取 canonical 保留 +历史,而不是压缩的活动 Todo 列表。归档的 open/deferred 行不能调度或提供新 +observation attribution;缺失或不匹配的历史、失效输入仍不能证明当前覆盖。 +历史证据不授予当前 claim、lease 或执行权限。 + +`status --goal-id research-demo --agent-id research-agent` 在 +`project_asset.bounded_research_frontier` 展示同一 live count 与有界 gap。 +`explore summary --goal-id research-demo --agent-id research-agent` 提供 +`research_execution_frontier`,并注释既有 node summary;相同 summary 进入 Lark +投影字段。`feishu-sync` 与 `feishu-card` 支持同一 Agent selector。多个已注册 +Agent 的 live-policy Goal 需要显式 selector;单个 Agent 可自动用于展示。 +这些读取不开始 Turn、不修改 Todo、不授予远程写权限;未激活策略保留原投影。 + +只关闭新策略、保留既有 planner: + +```sh +loopx configure-goal --goal-id research-demo --explore-composition-mode disabled --execute +``` + +移除策略字段也恢复既有行为。范围、输入或激活变化不能追溯抹去已准入 Turn 的 +guard;过期 writeback 仍被拒绝。当前 source fact 证明 invalidation 时,拒绝结果 +提供精确 `retirement_contract`。在原 `--replan-obligation-id` 和 +`--turn-instance-id` 下重用其 blocked progress 字段,并指定 +`--delivery-outcome outcome_gap`。Capability owner 重新校验当前源并生成 +`capability_obligation_retirement_v0`;调用者自报 approval flag 或任意 +blocker/evidence id 不能代替此证明。仍有绑定该 duty 的 runnable Todo 时不能 +retire,须先通过该任务自己的 lifecycle 暂停。无效、缺失或不可用的 evidence +source 不能冒充 invalidation。 + +通用 settlement owner 校验原 guard、durable writeback、当前 revision 与精确 +progress fingerprint,返回 `closeout_kind=capability_duty_retired_no_spend`。 +它仅关闭该已准入 Turn,不完成 Todo/Goal,也不花费 quota;已提交 debit 始终 +保留为普通历史 debit。旧 Turn reentry 与 spend 请求复用同一无支出 closeout, +不追加记账。下一轮重新判断当前 frontier,保留未完成 acceptance 与暂停任务。 +保留原 receipt 与证据,不通过删除它们制造结算。 +任何策略设置都不授权 Goal termination、真实模型 qualification、benchmark launch、 +部署或发布。 ## 停用与权限边界 From 3bf0dbe76b181a2e87341471d421288b0f16cc37 Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Mon, 28 Sep 2026 22:31:07 -0700 Subject: [PATCH 04/12] fix(explore): register completion evidence registry read Signed-off-by: Lihua <1017343802@qq.com> --- loopx/semantics/project_registry_io_manifest_v1.json | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index c3d04febe9..09ffa0406e 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -165,6 +165,14 @@ "api": "load_registry", "classification": "codec_api" }, + { + "site": "loopx/capabilities/explore/research_frontier.py::.hold_research_completion_evidence::codec_read:load_registry#1", + "line": 199, + "column": 31, + "kind": "codec_read", + "api": "load_registry", + "classification": "codec_api" + }, { "site": "loopx/capabilities/issue_fix/explore_projection.py::.project_issue_fix_explore_graph::codec_read:load_registry#1", "line": 643, From 14650fdd194eb713cf3d0627ae6e8b96d16b8ba2 Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Mon, 28 Sep 2026 22:38:04 -0700 Subject: [PATCH 05/12] chore(validation): refresh integrated registry census coordinates Signed-off-by: Lihua <1017343802@qq.com> --- loopx/semantics/project_registry_io_manifest_v1.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index f126b92447..2841bcae5c 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -495,7 +495,7 @@ }, { "site": "loopx/cli.py::.main::codec_read:load_project_registry#1", - "line": 846, + "line": 835, "column": 17, "kind": "codec_read", "api": "load_project_registry", From 85c795807b0077c3582c62b81c56ba5cafc352c1 Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Mon, 28 Sep 2026 22:51:35 -0700 Subject: [PATCH 06/12] fix(explore): qualify execution against current live lineage Signed-off-by: Lihua <1017343802@qq.com> --- .../capabilities/explore/research_evidence.py | 30 +++++++++++++------ .../explore_research_execution.ts | 17 +++++++++-- .../project_registry_io_manifest_v1.json | 8 +++++ 3 files changed, 44 insertions(+), 11 deletions(-) diff --git a/loopx/capabilities/explore/research_evidence.py b/loopx/capabilities/explore/research_evidence.py index 6cff3c9703..59345d6de8 100644 --- a/loopx/capabilities/explore/research_evidence.py +++ b/loopx/capabilities/explore/research_evidence.py @@ -105,21 +105,33 @@ def append_research_observation( if registry_path is None or runtime_root is None: raise ValueError("research execution lineage requires the selected registry and source runtime") from ...todos import list_goal_todos + from ...history import load_registry + from ...materials import find_registry_goal + from .research_frontier import build_research_composition_frontier, read_research_todo_history lineage = canonical["execution_lineage"] - context = list_goal_todos( - registry_path=registry_path, runtime_root_arg=str(runtime_root), goal_id=goal_id, - role="agent", todo_id=lineage["successor_todo_id"], - ) - todos = context.get("todos") or [] + goal = find_registry_goal(load_registry(registry_path), goal_id) or {} + harness = (goal.get("spawn_policy") or {}).get("explore_harness") or {} + policy = _research_result("explore.research.composition_policy", {"harness": harness}) + candidate_sources = [{"node_id": event["result_id"], "research_observation": event["research_observation"]} + for event in events if event.get("research_observation")] + frontier = None + if policy["enabled"]: + history = read_research_todo_history(runtime_root=runtime_root, goal=goal) + frontier = build_research_composition_frontier(projection, candidate_sources=candidate_sources, + harness=harness, todos=history, agent_id=agent_id) + todos = [todo for todo in history if todo["todo_id"] == lineage["successor_todo_id"]] + else: + context = list_goal_todos( + registry_path=registry_path, runtime_root_arg=str(runtime_root), goal_id=goal_id, + role="agent", todo_id=lineage["successor_todo_id"], + ) + todos = context.get("todos") or [] todo = todos[0] if len(todos) == 1 else {} _research_result("explore.research.validate_execution", { "goal_id": goal_id, "agent_id": agent_id, "observation": canonical, "nodes": projection["nodes"], "edges": projection["edges"], - "candidate_sources": [ - {"node_id": event["result_id"], "research_observation": event["research_observation"]} - for event in events if event.get("research_observation") - ], + "candidate_sources": candidate_sources, "frontier": frontier, "todo": _research_todo_facts(todo), }) event = build_explore_node_event( diff --git a/loopx/control_plane/capabilities/explore_research_execution.ts b/loopx/control_plane/capabilities/explore_research_execution.ts index 390584d30a..93fa859087 100644 --- a/loopx/control_plane/capabilities/explore_research_execution.ts +++ b/loopx/control_plane/capabilities/explore_research_execution.ts @@ -287,10 +287,23 @@ export function validateResearchExecution(params: JsonObject): JsonObject { const lineage = requireJsonObject(observation.execution_lineage, "execution_lineage"); const todo = requireJsonObject(params.todo, "canonical execution Todo"); const node = (params.nodes as JsonObject[]).find(row => row.node_id === observation.explore_node_id); - const gap = researchCompositionGaps(params).find(row => row.gap_id === lineage.gap_id); const reject = (reason: string): never => { throw new EffectRuntimeRequestError(`research execution ${lineage.replan_obligation_id}: ${reason}; read the current Todo and Explore summary before explore observe`); }; + const frontier = params.frontier == null ? null : requireJsonObject(params.frontier, "execution frontier"); + if (frontier && (frontier.schema_version !== "research_composition_frontier_v0" + || frontier.goal_id !== params.goal_id || frontier.agent_id !== params.agent_id)) { + reject("the current execution frontier must belong to this Goal and actor"); + } + // M2 diagnostics retain their cold meaning. M3 uses the same current + // canonical lineage join as status and closeout, including retained Todos. + const live = frontier?.enabled === true; + const gap = (live ? array(frontier!.lineage_gaps) : researchCompositionGaps(params)) + .find(row => row.gap_id === lineage.gap_id); + const uncovered = live ? !!gap && ["pending", "scheduled"].includes(String(gap.status)) + && gap.obligation_id === lineage.replan_obligation_id + && object(frontier!.policy).coverage_scope_id === object(observation.progress).coverage_scope_id + : gap?.state === "pending"; if (lineage.goal_id !== params.goal_id || lineage.agent_id !== params.agent_id) { reject("Goal or actor differs from execution lineage"); } @@ -308,7 +321,7 @@ export function validateResearchExecution(params: JsonObject): JsonObject { reject("Todo must bind exactly this experiment node"); } if (node?.node_kind !== "experiment" || (node.agent_id && node.agent_id !== lineage.agent_id) - || !gap || gap.state !== "pending" + || !gap || !uncovered || !(gap.experiment_node_ids as string[]).includes(String(observation.explore_node_id)) || JSON.stringify(gap.input_observations) !== JSON.stringify(observation.input_observations)) { reject("the experiment must cover this pending gap's current exact input observations"); diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index 2841bcae5c..c62706a357 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -165,6 +165,14 @@ "api": "load_registry", "classification": "codec_api" }, + { + "site": "loopx/capabilities/explore/research_evidence.py::.append_research_observation::codec_read:load_registry#1", + "line": 113, + "column": 39, + "kind": "codec_read", + "api": "load_registry", + "classification": "codec_api" + }, { "site": "loopx/capabilities/explore/research_frontier.py::.hold_research_completion_evidence::codec_read:load_registry#1", "line": 199, From 8a16deb1c85d689ce704a86cc0fa57dafc762ccf Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Mon, 28 Sep 2026 22:51:35 -0700 Subject: [PATCH 07/12] test(explore): cover diagnostic history before bound execution Signed-off-by: Lihua <1017343802@qq.com> --- .../test_research_execution_authority.py | 53 +++++++++++++++++-- .../explore_research_execution.test.ts | 34 ++++++++++++ 2 files changed, 83 insertions(+), 4 deletions(-) diff --git a/tests/control_plane/test_research_execution_authority.py b/tests/control_plane/test_research_execution_authority.py index b4c7cbed03..409afe327b 100644 --- a/tests/control_plane/test_research_execution_authority.py +++ b/tests/control_plane/test_research_execution_authority.py @@ -2,6 +2,8 @@ from __future__ import annotations import json +import subprocess +import sys from pathlib import Path import pytest @@ -97,8 +99,9 @@ def test_execution_attribution_reads_promoted_provider( @pytest.mark.parametrize("provider", ["file", "sqlite"]) @pytest.mark.parametrize("resolution", ["observed", "dismissed", "deferred"]) +@pytest.mark.parametrize("diagnostic_timing", [None, "before_activation", "after_activation"]) def test_native_completion_refuses_missing_result_and_accepts_exact_observation( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, resolution: str, + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, resolution: str, diagnostic_timing: str | None, ) -> None: isolate_sqlite_runtime(tmp_path, monkeypatch) try: @@ -109,9 +112,22 @@ def test_native_completion_refuses_missing_result_and_accepts_exact_observation( harness = {"enabled": True, "composition_mode": "explicit_only", "composition_scope_id": "scope-joint"} registry.write_text(json.dumps({"common_runtime_root": str(runtime), "goals": [{ "id": goal, "repo": str(tmp_path), "state_file": state.name, "status": "active", - "spawn_policy": {"explore_harness": harness}, + "spawn_policy": {"explore_harness": {"enabled": True} if diagnostic_timing == "before_activation" else harness}, "coordination": {"agent_model": "peer_v1", "registered_agents": [agent]}, }]})) + def cli(*args: str, success: bool = True) -> dict: + process = subprocess.run([sys.executable, "-m", "loopx.entrypoint", "--format", "json", + "--registry", str(registry), "--runtime-root", str(runtime), *args], + cwd=Path(__file__).resolve().parents[2], capture_output=True, text=True, timeout=60) + assert (process.returncode == 0) is success, process.stdout + process.stderr + return json.loads(process.stdout) + + packet = tmp_path / "observation.json" + def record(value: dict, *, success: bool = True) -> dict: + packet.write_text(json.dumps(value)) + return cli("explore", "observe", "--goal-id", goal, "--agent-id", agent, + "--observation-json", str(packet), success=success) + log = explore_result_log_path(runtime, goal) for name in ["a", "b", "joint"]: append_explore_result_event(log, build_explore_node_event( @@ -126,6 +142,19 @@ def test_native_completion_refuses_missing_result_and_accepts_exact_observation( for name in ["a", "b"]: append_explore_result_event(log, build_explore_edge_event( goal_id=goal, from_node="joint", to_node=name, edge_type="depends_on")) + if diagnostic_timing: + append_explore_result_event(log, build_explore_node_event(goal_id=goal, node_id="diagnostic", + title="Retained diagnostic", node_kind="experiment", status="resolved")) + for name in ["a", "b"]: + append_explore_result_event(log, build_explore_edge_event(goal_id=goal, + from_node="diagnostic", to_node=name, edge_type="depends_on")) + cold = build_explore_result_projection(load_explore_result_events_strict(log, goal_id=goal), goal_id=goal) + diagnostic = observation("diagnostic") + diagnostic["input_observations"] = cold["research_frontier"]["gaps"][0]["input_observations"] + assert record(diagnostic)["written"] + if diagnostic_timing == "before_activation": + assert cli("configure-goal", "--goal-id", goal, "--explore-composition-mode", "explicit_only", + "--explore-composition-scope-id", "scope-joint", "--execute")["ok"] events = load_explore_result_events_strict(log, goal_id=goal) projection = build_explore_result_projection(events, goal_id=goal) frontier = build_research_composition_frontier(projection, @@ -133,6 +162,10 @@ def test_native_completion_refuses_missing_result_and_accepts_exact_observation( for e in events if e.get("research_observation")], harness=harness, todos=[], agent_id=agent) gap = frontier["selected_gap"] + assert gap["status"] == "pending" + if diagnostic_timing: + assert projection["research_frontier"]["observed_count"] == 1 + assert frontier["observed_count"] == 0 task = {"schema_version": "todo_item_v0", "todo_id": "todo_joint", "index": 1, "role": "agent", "status": "open", "done": False, "text": "Run the bounded joint experiment.", "archive_state": "active", "source_section": "Agent Todo", "priority": "P1", @@ -140,6 +173,8 @@ def test_native_completion_refuses_missing_result_and_accepts_exact_observation( "target_key": "joint", "explore_result_node_refs": ["joint"], "replan_obligation_id": gap["obligation_id"]} initialize_canonical_authority(runtime, goal, build_todo_runtime_shadow_projection(goal_id=goal, todos=[task]), state_path=state, provider=provider) + live = cli("explore", "summary", "--goal-id", goal, "--agent-id", agent)["research_execution_frontier"] + assert live["scheduled_count"] == 1 and live["observed_count"] == 0 from loopx.control_plane.work_items.task_lease import acquire_task_lease acquired = acquire_task_lease(registry_path=registry, runtime_root=runtime, goal_id=goal, todo_id="todo_joint", owner=agent, idempotency_key="native-research-proof", ttl_seconds=300) @@ -193,8 +228,16 @@ def capture_native(method, params, **kwargs): result["closure_basis"] = None result["composition_resolution"] = {"schema_version": "research_composition_resolution_v0", "disposition": "deferred", "evidence_ids": ["ev-joint"]} - recorded = append_research_observation(log, goal_id=goal, observation=result, agent_id=agent, - registry_path=registry, runtime_root=runtime) + recorded = record(result) + written = log.read_bytes() + replay = record(result) + assert replay["replayed"] and not replay["written"] + assert log.read_bytes() == written + if resolution != "deferred": + changed = json.loads(json.dumps(result)) + changed["progress"]["evidence_ids"].append("ev-new") + assert "pending gap" in json.dumps(record(changed, success=False)) + assert log.read_bytes() == written if resolution == "deferred": from loopx.todos import update_goal_todo from loopx.capabilities.explore.composition_frontier import project_live_explore_composition_frontier @@ -265,6 +308,8 @@ def live(): agent_id=agent, status_payload={"run_history": {"goals": json.loads(registry.read_text())["goals"]}}) assert live()[f"{resolution}_count"] == 1 assert live()["scheduled_count"] == 0 + cli_view = cli("explore", "summary", "--goal-id", goal, "--agent-id", agent)["research_execution_frontier"] + assert cli_view[f"{resolution}_count"] == 1 and cli_view["scheduled_count"] == 0 append_explore_result_event(log, build_explore_node_event(goal_id=goal, node_id="a", title="Research a", status="open")) assert live()["observed_count"] == 0 replay = complete_goal_todo(registry_path=registry, goal_id=goal, todo_id="todo_joint", diff --git a/tests/control_plane_ts/explore_research_execution.test.ts b/tests/control_plane_ts/explore_research_execution.test.ts index 591e6c6000..39eb0fdb92 100644 --- a/tests/control_plane_ts/explore_research_execution.test.ts +++ b/tests/control_plane_ts/explore_research_execution.test.ts @@ -285,6 +285,40 @@ test("stale fingerprints, reads, malformed lineage and already-observed inputs f assert.throws(() => validateResearchExecution(params), /pending gap/); }); +test("cold terminal diagnostics cannot discharge or block a live execution duty", () => { + const params = liveFixture(), nodes = params.nodes as JsonObject[], raw = params.observation as JsonObject; + const diagnostic: JsonObject = {...raw, explore_node_id: "diagnostic", + progress: {...raw.progress as JsonObject, work_item_id: "todo_diagnostic"}}; + delete diagnostic.execution_lineage; + params.nodes = [...nodes, {node_id: "diagnostic", node_kind: "experiment", status: "resolved", + research_observation: normalizeResearchObservation({observation: diagnostic})}]; + params.edges = [...params.edges as JsonObject[], ...["a", "b"].map(to_node => + ({from_node: "diagnostic", to_node, edge_type: "depends_on"}))]; + params.todos = [params.todo]; + assert.equal(researchCompositionGaps(params)[0].state, "observed"); + const scheduled = projectResearchComposition(params); + assert.equal(scheduled.scheduled_count, 1); + assert.equal(scheduled.observed_count, 0); + assert.doesNotThrow(() => validateResearchExecution({...params, frontier: scheduled})); + assert.throws(() => validateResearchExecution({...params, frontier: {...scheduled, agent_id: "another-agent"}}), /Goal and actor/); + assert.throws(() => validateResearchExecution({...params, frontier: {...scheduled, + policy: {...scheduled.policy as JsonObject, coverage_scope_id: "other-scope"}}}), /pending gap/); + params.nodes = [...(params.nodes as JsonObject[]).filter(node => node.node_id !== "joint"), + {...nodes[2], status: "resolved", research_observation: normalizeResearchObservation({observation: raw})}]; + const observed = projectResearchComposition(params); + assert.equal(observed.observed_count, 1); + assert.throws(() => validateResearchExecution({...params, frontier: observed}), /pending gap/); + const retained = {...params.todo as JsonObject, status: "done", claimed_by: null, archive_state: "archive", actionable_open: false}; + const retry = {...params.todo as JsonObject, todo_id: "todo_retry", explore_result_node_refs: ["retry"], target_key: "retry"}; + params.nodes = [...params.nodes as JsonObject[], {node_id: "retry", node_kind: "experiment", status: "resolved"}]; + params.edges = [...params.edges as JsonObject[], ...["a", "b"].map(to_node => + ({from_node: "retry", to_node, edge_type: "depends_on"}))]; + const retryRaw = {...raw, explore_node_id: "retry", progress: {...raw.progress as JsonObject, work_item_id: "todo_retry"}, + execution_lineage: {...raw.execution_lineage as JsonObject, successor_todo_id: "todo_retry"}}; + assert.throws(() => validateResearchExecution({...params, todo: retry, observation: retryRaw, + frontier: projectResearchComposition({...params, todos: [retained, retry]})}), /pending gap/); +}); + test("the three-card presentation budget cannot hide execution attribution", () => { const params = fixture(), nodes = params.nodes as JsonObject[]; const template = nodes[0].research_observation as JsonObject; From 163669713b2e14d6bebab8a47ed5ebee01599edb Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Mon, 28 Sep 2026 22:51:36 -0700 Subject: [PATCH 08/12] docs(explore): reconcile implemented composition boundary Signed-off-by: Lihua <1017343802@qq.com> --- docs/reference/protocols/research-observation-v0.md | 4 +++- docs/reference/protocols/research-observation-v0.zh-CN.md | 3 ++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/docs/reference/protocols/research-observation-v0.md b/docs/reference/protocols/research-observation-v0.md index f7b01dd247..f6b5369162 100644 --- a/docs/reference/protocols/research-observation-v0.md +++ b/docs/reference/protocols/research-observation-v0.md @@ -5,7 +5,9 @@ Explore owns this optional evidence contract. Its first consumer is node summary render the same derived facts. This is the M1 evidence substrate and M2 **read-only shadow**. The M3 development boundary adds optional execution attribution, an explicit-only live replan gate, and legacy/native File/SQLite -Todo closeout checks. Retirement/resumption and full status/Explore presentation remain open. +Todo closeout checks, source-qualified duty retirement, lease-fenced resumption, +and shared status/Explore presentation. Maintainer-reviewed integration and live +model/scientific or remote Lark qualification remain open. ## Record and read back diff --git a/docs/reference/protocols/research-observation-v0.zh-CN.md b/docs/reference/protocols/research-observation-v0.zh-CN.md index d7e0b64d26..a22fb2279f 100644 --- a/docs/reference/protocols/research-observation-v0.zh-CN.md +++ b/docs/reference/protocols/research-observation-v0.zh-CN.md @@ -4,7 +4,8 @@ Explore 拥有这一可选证据契约。第一个写入入口是 `loopx explore `loopx explore summary` 与现有 Lark Explore 节点摘要显示同一派生事实。 本批交付 M1 证据基础和 M2 **只读 shadow**。M3 开发边界增加可选执行 attribution 与 explicit-only live replan 门禁,以及 legacy/native File/SQLite Todo closeout -校验。Retirement/resumption 及完整 status/Explore 展示仍未完成。 +校验、来源限定的义务退役、lease-fenced 恢复和共享 status/Explore 展示。 +Maintainer review 集成与现场模型/科研、远端 Lark qualification 仍待完成。 ## 录入与读回 From fed89c20f368889a7f3464ed7a4405723a8b14a9 Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Tue, 29 Sep 2026 16:32:14 -0700 Subject: [PATCH 09/12] test: preserve admitted required-read commands in interaction smoke Signed-off-by: Lihua <1017343802@qq.com> --- .../control_plane/interaction-contract-state-machine-smoke.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/examples/control_plane/interaction-contract-state-machine-smoke.py b/examples/control_plane/interaction-contract-state-machine-smoke.py index b4fc695a97..1ae3ea1aba 100644 --- a/examples/control_plane/interaction-contract-state-machine-smoke.py +++ b/examples/control_plane/interaction-contract-state-machine-smoke.py @@ -607,7 +607,9 @@ def assert_required_reads_are_mirrored_into_execution_channels() -> None: expected = [ { "kind": "agent_scoped_evidence_log", - "command": "loopx evidence-log --goal-id interaction-state-machine-goal", + # Execution channels retain the admitted command verbatim. Trimming + # belongs to display compaction and can change quoted arguments. + "command": " loopx evidence-log --goal-id interaction-state-machine-goal ", } ] assert contract["agent_channel"]["required_reads"] == expected, contract From fdc2b80f74cedd530285bd912e6b016b75da6fcf Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Tue, 29 Sep 2026 19:14:37 -0700 Subject: [PATCH 10/12] test: track generated twins and turn-start hook projection Signed-off-by: Lihua <1017343802@qq.com> --- tests/architecture/test_turn_contract_generation.py | 7 ++++++- tests/control_plane/test_prompt_upgrade_hook.py | 10 +++++++++- 2 files changed, 15 insertions(+), 2 deletions(-) diff --git a/tests/architecture/test_turn_contract_generation.py b/tests/architecture/test_turn_contract_generation.py index e14adf24c5..cdb4b43697 100644 --- a/tests/architecture/test_turn_contract_generation.py +++ b/tests/architecture/test_turn_contract_generation.py @@ -260,7 +260,12 @@ def test_new_independent_twin_cannot_hide_behind_generated_pair(monkeypatch): ) assert counts is not None raw, generated, maintained, budget = map(int, counts.groups()) - assert generated == 1 and raw == maintained + generated + # Both reviewed generators now contribute one verified Python/TS pair. + from scripts.generate_semantic_bindings import verified_generated_paths as semantic_generated_paths + + assert "loopx/control_plane/turn_driver/turn_contract_generated.py" in generator.verified_generated_paths() + assert "loopx/control_plane/content_digest.py" in semantic_generated_paths() + assert generated == 2 and raw == maintained + generated from loopx.semantics.inventory import SourceFile target = smoke["check_dual_runtime_twins"].__globals__ diff --git a/tests/control_plane/test_prompt_upgrade_hook.py b/tests/control_plane/test_prompt_upgrade_hook.py index a8ade637cb..1a285cc132 100644 --- a/tests/control_plane/test_prompt_upgrade_hook.py +++ b/tests/control_plane/test_prompt_upgrade_hook.py @@ -158,6 +158,13 @@ def test_live_decision_adds_only_existing_required_read_channel(tmp_path, monkey receipt.write_bytes(contents) pending = build_live_quota_should_run_decision(status, **kwargs) assert pending["required_reads"][-1]["kind"] == "automation_prompt_upgrade" + assert "turn_start_capability_hook_dispatch" not in baseline + dispatch = pending["turn_start_capability_hook_dispatch"] + assert set(dispatch) == {"required_reads"} + assert len(dispatch["required_reads"]) == 1 + assert dispatch["required_reads"][0]["kind"] == pending["required_reads"][-1]["kind"] + assert dispatch["required_reads"][0]["command"] == pending["required_reads"][-1]["command"] + assert pending["required_reads"][-1]["source"] == "turn_start_capability_hook" assert pending["interaction_contract"]["agent_channel"]["required_reads"] == pending["required_reads"] hint = pending["required_reads"][-1] assert len(hint["command"]) > 360 @@ -168,7 +175,8 @@ def test_live_decision_adds_only_existing_required_read_channel(tmp_path, monkey assert envelope["compaction"]["hook_prompt_budget_bytes"] == 1536 assert build_turn_envelope(baseline)["compaction"]["budget_bytes"] == 8192 for key in baseline.keys() | pending.keys(): - if key not in {"required_reads", "interaction_contract", "protocol_action_packet"}: + if key not in {"required_reads", "interaction_contract", "protocol_action_packet", + "turn_start_capability_hook_dispatch"}: assert pending.get(key) == baseline.get(key), key _set_fixture_prompt(path, database, desired) assert build_live_quota_should_run_decision(status, **kwargs) == baseline From 6395df7222af0d06c72c67f64a41053d72b4ab12 Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Wed, 30 Sep 2026 15:30:27 -0700 Subject: [PATCH 11/12] test(runtime): align source-read and disclosure fixtures Include the selected Python adapter in the independent fingerprint oracle, sequence churn injection after the first concurrent read, and expect the current usage notice version. Restore exact-head Python CI assertions without changing runtime or test budgets. Signed-off-by: Lihua <1017343802@qq.com> --- .../control_plane/test_runtime_source_read_batching.py | 5 ++++- tests/control_plane/test_source_cli_entrypoint.py | 2 +- .../test_turn_journal_runtime_readiness.py | 10 ++++++++++ 3 files changed, 15 insertions(+), 2 deletions(-) diff --git a/tests/control_plane/test_runtime_source_read_batching.py b/tests/control_plane/test_runtime_source_read_batching.py index 1cd1f167c6..61122c68f2 100644 --- a/tests/control_plane/test_runtime_source_read_batching.py +++ b/tests/control_plane/test_runtime_source_read_batching.py @@ -1,6 +1,7 @@ from __future__ import annotations import hashlib +import sys from pathlib import Path import pytest @@ -9,8 +10,10 @@ def serial_fingerprint(root: Path) -> str: - """Independent reference: names and raw bytes, not decoded source text.""" + """Independent reference: selected Python adapter, names, and raw bytes.""" digest = hashlib.sha256() + digest.update(sys.executable.encode("utf-8")) + digest.update(str(sys.version_info[:3]).encode("ascii")) for path in sorted(p for p in root.rglob("*") if p.suffix in {".ts", ".json"}): digest.update(path.relative_to(root).as_posix().encode("utf-8")) digest.update(path.read_bytes()) diff --git a/tests/control_plane/test_source_cli_entrypoint.py b/tests/control_plane/test_source_cli_entrypoint.py index fdb5fcd150..c0cce900d5 100644 --- a/tests/control_plane/test_source_cli_entrypoint.py +++ b/tests/control_plane/test_source_cli_entrypoint.py @@ -311,7 +311,7 @@ def test_source_first_usage_disclosure_keeps_json_pure_and_does_not_send(tmp_pat assert json.loads(result.stdout)["ok"] is True assert "random installation ID" in result.stderr stored = json.loads((state / "usage-ping.json").read_text()) - assert stored["notice"]["version"] == 4 + assert stored["notice"]["version"] == 5 assert "last_attempt_day" not in stored and "counters" not in stored diff --git a/tests/control_plane/test_turn_journal_runtime_readiness.py b/tests/control_plane/test_turn_journal_runtime_readiness.py index cb0a2b0165..3b3b305b21 100644 --- a/tests/control_plane/test_turn_journal_runtime_readiness.py +++ b/tests/control_plane/test_turn_journal_runtime_readiness.py @@ -1,5 +1,6 @@ from __future__ import annotations +from threading import Event from pathlib import Path from typing import Any @@ -117,12 +118,16 @@ def test_runtime_fingerprint_rescans_when_a_snapshotted_file_disappears_while_re later.write_text("export const later = true;\n", encoding="utf-8") original_read_bytes = Path.read_bytes reads: list[str] = [] + first_read_done = Event() def remove_later_after_first_read(path: Path) -> bytes: + if path == later: + assert first_read_done.wait(5), "first source read did not finish" reads.append(path.name) content = original_read_bytes(path) if path == first and later.exists(): later.unlink() + first_read_done.set() return content monkeypatch.setattr(effect_runtime, "_control_plane_root", lambda: tmp_path) @@ -142,17 +147,22 @@ def _install_persistent_stat_read_churn( original_scan = effect_runtime._scan_runtime_source_files original_read_bytes = Path.read_bytes scans: list[tuple[str, ...]] = [] + first_read_done = Event() def restore_then_scan(root: Path) -> tuple[str, ...]: + first_read_done.clear() later.write_text("export const later = true;\n", encoding="utf-8") files = original_scan(root) scans.append(files) return files def remove_later_after_first_read(path: Path) -> bytes: + if path == later: + assert first_read_done.wait(5), "first source read did not finish" content = original_read_bytes(path) if path == first: later.unlink() + first_read_done.set() return content monkeypatch.setattr(effect_runtime, "_control_plane_root", lambda: tmp_path) From 5f9705d375877de85483e5056b0c3700789c65b7 Mon Sep 17 00:00:00 2001 From: Lihua <1017343802@qq.com> Date: Wed, 30 Sep 2026 16:41:29 -0700 Subject: [PATCH 12/12] test(coverage): exclude disposable probe checkout from shards The semantic inventory CLI fixture executes copied loopx sources in a temporary repository. Suppress pytest-cov subprocess instrumentation only for that fixture child, so combined shard coverage does not depend on files removed after the test. Keep checkout coverage and the existing floor intact. Signed-off-by: Lihua <1017343802@qq.com> --- tests/architecture/test_semantic_development_probe.py | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/tests/architecture/test_semantic_development_probe.py b/tests/architecture/test_semantic_development_probe.py index 94987444c4..ec49dc2ef1 100644 --- a/tests/architecture/test_semantic_development_probe.py +++ b/tests/architecture/test_semantic_development_probe.py @@ -2,6 +2,7 @@ from __future__ import annotations +import os from pathlib import Path import subprocess import sys @@ -344,6 +345,13 @@ def probe_cli(repository: Path) -> Path: def _run_probe_cli(repository: Path) -> subprocess.CompletedProcess[str]: + # The child executes copied sources under a disposable loopx/ package. Do + # not attribute those temporary files to the checkout's coverage artifact: + # shard artifacts are combined after pytest has removed the temp checkout. + env = { + key: value for key, value in os.environ.items() + if not key.startswith("COV_CORE_") and key != "COVERAGE_PROCESS_START" + } return subprocess.run( [ sys.executable, @@ -352,6 +360,7 @@ def _run_probe_cli(repository: Path) -> subprocess.CompletedProcess[str]: "HEAD", ], cwd=repository, + env=env, capture_output=True, text=True, check=False,