Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 77 additions & 0 deletions .github/scripts/report_pytest_failures.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,77 @@
# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

"""Expose CI failure locations without publishing assertion payloads or secrets."""

import importlib.metadata
from pathlib import Path
import re
import sys
import xml.etree.ElementTree as ET


def failure_locations(report: Path) -> list[str]:
if not report.exists():
return ["JUnit report unavailable; check the preceding installation/test step."]
locations = []
for case in ET.parse(report).getroot().iter("testcase"):
failure = next((x for x in case if x.tag in {"failure", "error"}), None)
if failure is None:
continue
# Test parameters and assertion text can contain credentials or prompts.
# Publish only Python identifiers and repository test source locations.
name = case.get("name", "").split("[", 1)[0]
name = name if re.fullmatch(r"[a-zA-Z_][a-zA-Z_0-9]*", name) else "collection"
classname = case.get("classname", "")
classname = (
classname
if re.fullmatch(r"[a-zA-Z_][a-zA-Z_0-9.]*", classname)
else "tests"
)
source = re.findall(
r"^(tests/[a-zA-Z_0-9/]+\.py):(\d+):", failure.text or "", re.MULTILINE
)
suffix = f" at {source[-1][0]}:{source[-1][1]}" if source else ""
locations.append(f"{classname}.{name}{suffix}")
return locations


def main() -> None:
packages = [
"google-adk",
"litellm",
"google-genai",
"pydantic",
"pytest",
"pytest-asyncio",
"sqlalchemy",
"aiosqlite",
"agentkit-sdk-python",
"opentelemetry-sdk",
]
versions = []
for name in packages:
try:
version = importlib.metadata.version(name)
except importlib.metadata.PackageNotFoundError:
version = "missing"
if re.fullmatch(r"[a-zA-Z0-9._+!-]+", version):
versions.append(f"{name}={version}")
print("::notice title=Dependency versions::" + "; ".join(versions))
for location in failure_locations(Path(sys.argv[1]))[:30]:
print("::error title=Pytest failure location::" + location)


if __name__ == "__main__":
main()
81 changes: 81 additions & 0 deletions .github/workflows/context-compression-gate.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,81 @@
name: Context Compression Gate

on:
workflow_call:
pull_request:
paths:
- 'veadk/context/**'
- 'veadk/agent.py'
- 'veadk/runner.py'
- 'veadk/memory/**'
- 'veadk/agents/**'
- 'veadk/models/**'
- 'veadk/cli/generated_agent_*.py'
- 'veadk/cli/cli_frontend.py'
- 'veadk/integrations/agentkit/app.py'
- 'frontend/src/create/**'
- 'frontend/src/adk/client.ts'
- 'frontend/src/ui/AgentTopology.tsx'
- 'frontend/src/i18n/resources/**'
- 'frontend/tests/contextCompression*.test.mjs'
- 'veadk/extensions/harness/plugins/compactor/**'
- 'tests/context/**'
- 'tests/agent/**'
- 'tests/cli/test_generated_agent_backend_codegen*.py'
- 'evaluations/context_compression/**'
- 'tests/models/**'
- 'tests/run_context_compression_gate.py'
- 'tests/test_context_release_gate.py'
- 'tests/test_ci_failure_summary.py'
- '.github/scripts/report_pytest_failures.py'
- '.github/workflows/context-compression-gate.yaml'
- '.github/workflows/publish-tag-to-pypi.yaml'
- '.github/workflows/publish-studio-release.yaml'
- 'pyproject.toml'
- 'uv.lock'

permissions:
contents: read

jobs:
studio-contracts:
runs-on: ubuntu-latest
timeout-minutes: 10
defaults:
run:
working-directory: frontend
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: actions/setup-node@v4
with:
node-version: '22'
cache: npm
cache-dependency-path: frontend/package-lock.json
- run: npm ci --ignore-scripts --no-audit --no-fund
- run: node --test tests/contextCompression*.test.mjs
contracts:
runs-on: ubuntu-latest
timeout-minutes: 15
strategy:
fail-fast: false
matrix:
python: ['3.10', '3.12']
adk: ['1.34.0', '2.1.0', '2.2.0', '2.9.2']
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: actions/setup-python@v5
with:
python-version: ${{ matrix.python }}
- name: Install fixed ADK compatibility target
run: python -m pip install -e '.[dev]' 'google-adk==${{ matrix.adk }}'
- name: Check installed dependency consistency
run: python -m pip check
- name: Run isolated SDK and Runner contracts
run: python tests/run_context_compression_gate.py -q --junitxml="$GITHUB_WORKSPACE/context-contracts.xml"
- name: Report failure locations and dependency versions
if: failure()
run: python .github/scripts/report_pytest_failures.py context-contracts.xml
4 changes: 4 additions & 0 deletions .github/workflows/publish-studio-release.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -44,13 +44,17 @@ concurrency:
cancel-in-progress: ${{ github.event_name == 'pull_request' }}

jobs:
context-compression-gate:
uses: ./.github/workflows/context-compression-gate.yaml

harness-sidecar-release-gate:
if: >-
github.repository == 'volcengine/veadk-python' &&
github.event_name == 'pull_request'
uses: ./.github/workflows/harness-sidecar-release-gate.yaml

verify:
needs: context-compression-gate
if: >-
github.repository == 'volcengine/veadk-python' &&
(github.event_name == 'pull_request' || github.ref == 'refs/heads/main')
Expand Down
5 changes: 4 additions & 1 deletion .github/workflows/publish-tag-to-pypi.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5,11 +5,14 @@ on:
workflow_dispatch:

jobs:
context-compression-gate:
uses: ./.github/workflows/context-compression-gate.yaml

harness-sidecar-release-gate:
uses: ./.github/workflows/harness-sidecar-release-gate.yaml

build:
needs: harness-sidecar-release-gate
needs: [harness-sidecar-release-gate, context-compression-gate]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
Expand Down
10 changes: 9 additions & 1 deletion .github/workflows/unit-tests.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 30
strategy:
fail-fast: false
matrix:
python-version: ["3.10", "3.12"]

Expand Down Expand Up @@ -61,7 +62,14 @@ jobs:
# `codex_smoke` is excluded here rather than relying on its
# CODEX_RUN_SMOKE opt-in alone: it spawns a real Codex subprocess and
# binds two real loopback ports, which must never run under `-n 16`.
pytest -n 16 -m "not codex_smoke"
pytest -n 16 -m "not codex_smoke" --junitxml=unit-tests.xml

- name: Report failure locations and dependency versions
if: failure()
run: |
if [ -x .venv/bin/python ]; then
.venv/bin/python .github/scripts/report_pytest_failures.py unit-tests.xml
fi

# Real Codex binary + real OS sandbox + real shim socket, against a stubbed
# model backend (no credentials, no network egress). Kept out of the matrix
Expand Down
79 changes: 79 additions & 0 deletions docs/content/docs/framework/agent/context-compression.en.mdx
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
---
title: Context compression
description: Keep complete sessions and reduce model input with recoverable references
---

Standard `veadk.Agent` instances and Studio-generated LLM Agents enable recoverable
context compression by default. The default Runner stores full sessions in project
SQLite at `.adk/session.db`; configured session services remain in use. Compression
changes model input only. Selected evidence and source references replace older
material, and the model searches or pages through originals only when needed.

```python
from veadk import Agent, Runner

agent = Agent(name="assistant", context_compression={
"context_window": 64000,
"input_limit": 48000,
"output_reserve": 8000,
"trigger_ratio": 0.8,
"target_ratio": 0.6,
"summary_trigger_ratio": 0.95,
})
runner = Runner(agent=agent, app_name="my_app", user_id="user")
```

All fields are optional for known models. Custom capacities need documented window
and output settings. Effective input also respects model limits, output reserve and
safety margin. Ratios use this available budget and must satisfy
`0 < target_ratio < trigger_ratio <= summary_trigger_ratio <= 1`. They are targets,
not guaranteed savings. Conservative UTF-8 byte accounting may trigger earlier than
provider token usage. Irreducible protected input fails before an oversized request;
media requires a suitable `media_token_reserve` when its size cannot be estimated.
Output reserve plans space; it does not by itself cap generated output.

Reviewed capacities live in `veadk/context/model_capacity.py`, with provider/model
IDs, context/input/output ceilings, source URLs and verification dates. The SDK
checks this table first, then exact entries in the installed LiteLLM local catalogue.
It makes no capacity API calls or runtime downloads and never guesses by model family.
Ark's public model-version metadata does not yet establish the full capacity contract.
An explicit window may reduce, but cannot exceed, a known smaller model limit.

If neither local table covers the model, supply a verified `context_window` and an
output budget. Otherwise `ContextBudgetError` with `model_capacity_required` is raised
before sending, including for unknown `ep-*` deployments and when compression is off.
Unknown fallbacks use `fallback_capacity_required`; they cannot inherit the primary
model's window. `input_limit` alone does not establish deployment capacity. Status
queries remain available and report `needs_configuration` for a missing capacity.

The reviewed table uses a 16,384-token planning reserve; explicit generation limits
and Ark's answer/reasoning semantics may change the effective reserve. Where only
an input cap is published (Gemini), it also serves as a conservative shared ceiling;
input and output maxima are never added to invent a larger context window. Capacity
coverage does not claim live compatibility testing for every provider or modality.

Both Studio creation views expose these controls in model settings, with percentages
for ratios. Drafts, nested LLM Agents, YAML and generated Python preserve the policy.
Explicit `mode: "off"` remains off. Regenerate a project after changing its policy.

The SDK automatically binds a lazy hybrid retriever and indexes eligible sources
under pressure. Its disposable index lives at `.adk/context-index.sqlite3`. Ark Agents
reuse existing access only for the same official endpoint. Existing `MODEL_EMBEDDING_*`
settings configure model, dimension, base URL and key; other providers need explicit
embedding access. The default permits 64 embedding requests per Agent invocation,
4 concurrent requests, no retries, and a shared 5-second retrieval deadline. Unavailable
services or partial indexing use local evidence selection. Originals remain readable.
Embedding and occasional summaries add API usage. Short input opens no index or client.

Set `retrieval: "lexical"` to avoid embedding, or configure `index_path` and
`embedding_max_calls` in the policy. Explicit `use_context_retriever(...)` bindings
remain caller-owned; the default retriever closes after each invocation.

Use `context_compression=False` to disable transformations. Capacity admission still
applies, including rejection of unknown capacities. Inspect `agent.context_compression_status` for capacity gaps or unsupported
custom model adapters/external runtimes. These controls apply to the standard ADK
model loop; external `codex`/`piagent` runtimes and live audio/video own their loops.
SQLite requires persistent storage for restart recovery and is not shared across
instances. Use a shared Session database for multiple instances. Reads are scoped
to authorized original Session records, never authorized merely by an index entry.
Validate compressed-answer quality on representative business cases.
Loading
Loading