Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions docs/policy.md
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,11 @@ policy deny (the node does not run).
/ `memory.forget`. Memory output is untrusted; `tools.write_file.on_tainted:
deny` is how a poisoned recall is stopped. See [memory.md](memory.md).

`type: decide` is governed under `deciders.<name>` (for example `jev`, `shim`).
Rules may `on_tainted: deny` and pin `allow_models` (so an org can allow
`jev-1.13.0` and refuse `jev-latest`). `default: deny` without a matching
decider rule refuses the call. See [decisions.md](decisions.md).

`type: a2a` is governed as tool name `a2a`. `default: deny` without a
`tools.a2a` rule refuses delegation. `tools.a2a.allow_hosts` and
`egress.allow_hosts` restrict destination hosts. When a policy file exists,
Expand Down
6 changes: 6 additions & 0 deletions docs/security-model.md
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,12 @@ system prompt, decides whether a tool runs.
- Agent tool-calls whose prompt or system interpolates untrusted state, even
when the model emits literal arguments.
- Known secret values being placed into a model request.
- A `type: decide` node using untrusted state as the sole gate on an
irreversible action — **it must not be**. Jev is documented as vulnerable to
adversarial text injection. Confidence is a margin, not trustworthiness.
`min_confidence` is a quality control, not a security control. Decide output
is tainted from its state. Defence in depth, not a solution to prompt
injection. See [decisions.md](decisions.md).

## Supply chain

Expand Down
5 changes: 5 additions & 0 deletions docs/sovereign.md
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,11 @@ Allowlisted destinations and every connect attempt (permitted or refused) are
recorded on the run. A refusal raises `EgressDenied` naming destination and
node.

`type: decide` with `decider: jev` is a hosted call (no local weights) and is
refused **before the first node**, naming the decide node. `decider: shim` over
a local model (for example Ollama) is the sovereign-compatible path. See
[decisions.md](decisions.md).

## What the attestation proves

`readyagents attest RUN_ID` emits mode, recorded attempts, allowlist, model
Expand Down
111 changes: 106 additions & 5 deletions src/readyagents/decide/node.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@
from readyagents.decide.base import questions_from_mapping
from readyagents.decide.registry import get_decider
from readyagents.decide.types import Decision
from readyagents.errors import CassetteMiss, DecideError
from readyagents.errors import CassetteMiss, DecideError, EgressDenied, PolicyDenied
from readyagents.logging import get_logger
from readyagents.workflow.templates import interpolate, interpolate_value

Expand Down Expand Up @@ -66,23 +66,25 @@ def run_decide_node(node: Any, state: Any, ctx: Any) -> Any:
if callable(raise_if):
raise_if(run_id=getattr(state, "run_id", None))

judged = _redact_state(judged, ctx)
vendor_state = _redact_state(judged, ctx)

blob = json.dumps(judged, ensure_ascii=False, default=str)
tokens = heuristic_tokens(blob + json.dumps({k: q.wire() for k, q in questions.items()}))
decider, model_id = get_decider(
model_hint or None,
secrets=getattr(ctx, "secrets", None),
offline=False,
llm=getattr(ctx, "llm", None),
min_confidence=min_conf,
)
_enforce_decider_policy(ctx, node, state, decider.name, model_id)

blob = json.dumps(vendor_state, ensure_ascii=False, default=str)
tokens = heuristic_tokens(blob + json.dumps({k: q.wire() for k, q in questions.items()}))
meter = getattr(ctx, "spend_meter", None)
if meter is not None:
consult = getattr(meter, "consult_before_call", None)
if callable(consult):
consult(model_id, prompt_tokens=tokens)
decision = decider.decide(state=judged, questions=questions, model=model_id)
decision = decider.decide(state=vendor_state, questions=questions, model=model_id)
if meter is not None:
record_usage = getattr(meter, "record_usage", None)
if callable(record_usage):
Expand Down Expand Up @@ -138,9 +140,93 @@ def _finish(
payload["routed"] = nxt
payload["route_reason"] = reason
payload["next"] = nxt
_audit_decide(ctx, state, node, decision, judged, nxt, low)
return payload


def _audit_decide(
ctx: Any,
state: Any,
node: Any,
decision: Decision,
judged: Any,
nxt: str | None,
low: list[str],
) -> None:
auditor = getattr(ctx, "auditor", None)
if auditor is None:
return
import hashlib

from readyagents.llm.cache import canonical_json_bytes

state_hash = hashlib.sha256(canonical_json_bytes(judged)).hexdigest()
auditor(
"decide",
run_id=getattr(state, "run_id", None),
node_id=node.id,
decider=decision.decider,
model=decision.model,
question_keys=list(decision.answers),
routed=nxt,
low_confidence=low,
state_hash=state_hash,
)


def _enforce_decider_policy(
ctx: Any, node: Any, state: Any, decider_name: str, model_id: str
) -> None:
policy = getattr(ctx, "policy", None)
if policy is None or not hasattr(policy, "decider_rule"):
return
rule_id, rule = policy.decider_rule(decider_name)
if rule is None and getattr(policy, "default", "allow") == "deny":
raise PolicyDenied(node.id, f"decider '{decider_name}' is not allowed", rule="default")
if rule is not None and rule.allow_models:
allowed = {str(item) for item in rule.allow_models}
if model_id not in allowed and f"{decider_name}:{model_id}" not in allowed:
raise PolicyDenied(
node.id,
f"decider model '{model_id}' is not allowed",
rule=rule_id,
)
tainted = _state_is_tainted(node, state)
action = getattr(rule, "on_tainted", "allow") if rule is not None else "allow"
if tainted and action == "deny":
raise PolicyDenied(node.id, "tainted state cannot use this decider", rule=rule_id)


def preflight_sovereign_decide(workflow: Any, *, settings: Any) -> None:
"""Refuse hosted Jev before the first node when --sovereign is on."""
from readyagents.secrets import secret_for_provider
from readyagents.sovereign.egress import is_loopback_url

base = str(getattr(settings, "typesafe_base_url", None) or "https://api.typesafe.ai")
has_key = bool(secret_for_provider("typesafe", settings=settings))
for node in getattr(workflow, "nodes", None) or []:
kind = str(getattr(node, "type", "") or "")
if kind == "decide":
ref = str(getattr(node, "decider", None) or "")
elif kind == "classify":
rem = dict(getattr(node, "model_for_remainder", None) or {})
ref = str(rem.get("decider") or "")
else:
continue
name = ref.split(":", 1)[0].strip().lower() if ref else ""
if name == "shim":
continue
if name == "":
if kind != "decide" or not has_key:
continue
name = "jev"
if name != "jev":
continue
if is_loopback_url(base):
continue
raise EgressDenied(base, node_id=str(node.id))


def _route(
node: Any, decision: Decision, min_conf: float | None
) -> tuple[str | None, str, list[str]]:
Expand Down Expand Up @@ -181,6 +267,21 @@ def _route(
return getattr(node, "next", None), "next", low


def _state_is_tainted(node: Any, state: Any) -> bool:
from readyagents.firewall.taint import UNTRUSTED, provenance_for_template, walk_strings

raw = getattr(node, "state", None)
texts: list[str]
if isinstance(raw, str):
texts = [raw]
else:
texts = [text for text in walk_strings(raw) if "{{" in str(text)]
return any(
provenance_for_template(state, text, node_id=getattr(node, "id", None)).trust == UNTRUSTED
for text in texts
)


def _render_state(raw: Any, run_state: Any) -> Any:
ns = run_state.mapping()
if isinstance(raw, str):
Expand Down
21 changes: 21 additions & 0 deletions src/readyagents/firewall/policy_file.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,13 @@ class NodeRule(_Forbid):
require_approval: bool = False


class DeciderRule(_Forbid):
"""Allow/deny a System One decider by name; optionally pin model ids."""

on_tainted: Action = "allow"
allow_models: list[str] | None = None


class EgressBlock(_Forbid):
allow_hosts: list[str] | None = None

Expand All @@ -51,6 +58,7 @@ class Policy(_Forbid):
tools: dict[str, ToolRule] = Field(default_factory=dict)
detection: DetectionBlock | None = None
nodes: dict[str, NodeRule] = Field(default_factory=dict)
deciders: dict[str, DeciderRule] = Field(default_factory=dict)
require_signed: bool = False
frozen: bool = False
on_lock_mismatch: Action = "allow"
Expand All @@ -70,6 +78,19 @@ def tool_rule(self, name: str) -> tuple[str, ToolRule | None]:
_length, key, rule = matches[0]
return key, rule

def decider_rule(self, name: str) -> tuple[str, DeciderRule | None]:
if name in self.deciders:
return name, self.deciders[name]
matches: list[tuple[int, str, DeciderRule]] = []
for key, rule in self.deciders.items():
if _name_matches(key, name):
matches.append((len(key), key, rule))
if not matches:
return "default", None
matches.sort(reverse=True)
_length, key, rule = matches[0]
return key, rule


def _name_matches(pattern: str, name: str) -> bool:
from fnmatch import fnmatch
Expand Down
17 changes: 17 additions & 0 deletions src/readyagents/firewall/taint.py
Original file line number Diff line number Diff line change
Expand Up @@ -263,6 +263,23 @@ def note_node_output(state: RunState, node: Any, output: Any) -> None:
if isinstance(output, dict) and output.get("role") == "human_agent":
source = "human_agent"
prov = untrusted(source=source, node_id=node_id)
elif kind == "decide":
raw_state = getattr(node, "state", None)
if isinstance(raw_state, str):
prov = provenance_for_template(state, raw_state, node_id=node_id)
else:
parts = [
provenance_for_template(state, text, node_id=node_id)
for text in walk_strings(raw_state)
if "{{" in str(text)
]
prov = (
merge(parts, node_id=node_id)
if parts
else trusted(source="decide", node_id=node_id)
)
if prov.trust == UNTRUSTED:
prov = untrusted(source="decide", node_id=node_id)
elif kind == "browser":
url = "browser"
if isinstance(output, dict):
Expand Down
1 change: 1 addition & 0 deletions src/readyagents/replay/record.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,7 @@ def known_secret_values(settings: Any, secrets: Any = None) -> list[str]:
"aws_secret_access_key",
"aws_session_token",
"vertex_project",
"typesafe_api_key",
"decision_secret",
):
value = getattr(settings, attr, None) if settings is not None else None
Expand Down
3 changes: 3 additions & 0 deletions src/readyagents/workflow/runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -363,8 +363,11 @@ def run_workflow_file(
if str(item).strip()
]
if want_sovereign:
from readyagents.decide.node import preflight_sovereign_decide
from readyagents.sovereign.egress import install_guard

if not dry_run:
preflight_sovereign_decide(workflow, settings=settings)
guard = install_guard(allow)
if workflow.mcp_servers and not dry_run and not pending_lock_gate:
from readyagents.mcp.client import MCPClient
Expand Down
86 changes: 86 additions & 0 deletions tests/test_decide.py
Original file line number Diff line number Diff line change
Expand Up @@ -856,3 +856,89 @@ def test_decide_example_validates_and_dry_runs() -> None:
assert d1.exit_code == 0, d1.stdout + d1.stderr
d2 = runner.invoke(app, ["run", path, "--dry-run", "--no-persist"])
assert d2.exit_code == 0, d2.stdout + d2.stderr


def test_redactor_applies_to_vendor_body_not_cassette_digest(
tmp_settings, tmp_path: Path, monkeypatch
) -> None:
"""ctx.redactor mutates the request body; record/replay digest the pre-redact state."""
from readyagents.policy import REDACTED, Redactor
from readyagents.replay.cassette import Cassette
from readyagents.testing.helpers import run_workflow_spec

email = "ada@x.test"
captured: dict[str, Any] = {}

def exchange(url, *, method, body, headers, timeout):
captured["body"] = json.loads(body.decode("utf-8"))
payload = {
"model": "jev-1.13.0",
"answers": {
"department": {"type": "choice", "choice": "technical", "confidence": 0.9},
"is_urgent": {"type": "noul", "noul": 0.99, "confidence": 0.98},
},
"usage": {"input_tokens": 8, "output_tokens": 1},
}
return 200, json.dumps(payload).encode("utf-8"), {}

jev = JevDecider(SECRET, sleep=lambda _s: None)
monkeypatch.setattr("readyagents.decide.node.get_decider", lambda *a, **k: (jev, "jev-1.13.0"))
spec = _triage_spec()
spec["inputs"]["message"] = f"checkout is down, ping {email}"
tape = Cassette.new(run_id="r1", workflow="triage")
token = use_transport(exchange)
try:
first = run_workflow_spec(
spec,
pin_home=tmp_settings.home_path(),
workflow_dir=tmp_path,
cassette=tape,
recording=True,
redactor=Redactor(),
)
finally:
reset_transport(token)
assert first.output_keys["summary"] == "page"
sent = json.dumps(captured["body"])
assert email not in sent
assert REDACTED in sent

path = tmp_path / "c.json"
tape.save(path, root=tmp_path)
assert email not in path.read_text(encoding="utf-8")

def boom(*_a, **_k):
raise AssertionError("offline replay must not construct a decider")

monkeypatch.setattr("readyagents.decide.node.get_decider", boom)
loaded = Cassette.load(path)
second = run_workflow_spec(
spec,
pin_home=tmp_settings.home_path(),
workflow_dir=tmp_path,
cassette=loaded,
offline=True,
redactor=Redactor(),
)
assert second.output_keys["summary"] == "page"


def test_decide_cancellation_before_call(tmp_settings, tmp_path: Path, monkeypatch) -> None:
from readyagents.errors import CancellationRequested
from readyagents.testing.helpers import run_workflow_spec
from readyagents.workflow.cancellation import CancellationToken

token = CancellationToken()
token.request(reason="stop")

def getter(*_a, **_k):
raise AssertionError("cancellation must run before constructing a decider")

monkeypatch.setattr("readyagents.decide.node.get_decider", getter)
with pytest.raises(CancellationRequested):
run_workflow_spec(
_triage_spec(),
pin_home=tmp_settings.home_path(),
workflow_dir=tmp_path,
cancellation=token,
)
Loading
Loading