From b4269e3e9b2257139e87fc12ab7c2ac066ef948e Mon Sep 17 00:00:00 2001 From: Mihai Gheorghe Date: Wed, 7 Oct 2026 17:22:56 +0300 Subject: [PATCH 1/2] feat(agent): cap the agent's context window with agent.context_window Add a harness-neutral `agent.context_window` field (tokens) so a benchmark operator can run the same tasks at different context caps, per agent config or per experiment variant. - claude-code: sent as CLAUDE_CODE_AUTO_COMPACT_WINDOW in the CLI env, which outranks --autocompact and every settings scope; range 100000-1000000 (the CLI silently drops out-of-range settings values, so the range is enforced at load). Setting claude_settings.autoCompactWindow too is a load error. - codex: sent as model_context_window in the thread config; Codex caps it to the model's catalog maximum and auto-compacts at 90% of it. - every other agent type rejects the field at load instead of running uncapped under a capped label. The resolved value is recorded in agent_config on every task row and in the config lineage. Rationale: .claude/notes/context-window.md Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01RarvxSsr4K2KkBisTbXxQY --- .claude/notes/README.md | 1 + .claude/notes/context-window.md | 159 +++++++++++++++++++++ docs/AB_EXPERIMENTS.md | 24 ++++ docs/TASK_DEFINITION_GUIDE.md | 8 ++ docs/agents/CLAUDE_CODE.md | 1 + docs/agents/CODEX.md | 7 + docs/agents/HARNESS_PARITY.md | 24 ++++ src/coder_eval/agents/claude_code_agent.py | 10 ++ src/coder_eval/agents/codex_agent.py | 4 + src/coder_eval/models/agent_config.py | 76 ++++++++++ tests/test_agent_context_window.py | 121 ++++++++++++++++ 11 files changed, 435 insertions(+) create mode 100644 .claude/notes/context-window.md create mode 100644 tests/test_agent_context_window.py diff --git a/.claude/notes/README.md b/.claude/notes/README.md index 641e4f1db..e74a8b89e 100644 --- a/.claude/notes/README.md +++ b/.claude/notes/README.md @@ -13,6 +13,7 @@ rule docstrings in `tests/lint/rules/`, then the guides under `docs/`. ## Contents - [agents.md](agents.md) — agent adapters, the turn lifecycle, token reconciliation, harness parity +- [context-window.md](context-window.md) — the `agent.context_window` cap, per harness - [contracts.md](contracts.md) — criteria, datasets, aggregation, judging - [isolation.md](isolation.md) — the docker driver, the sandbox, detached grading - [lint-rules.md](lint-rules.md) — why each CE lint rule exists diff --git a/.claude/notes/context-window.md b/.claude/notes/context-window.md new file mode 100644 index 000000000..8086be927 --- /dev/null +++ b/.claude/notes/context-window.md @@ -0,0 +1,159 @@ +# Capping the agent's context window + +Analysis and design for `agent.context_window`: one harness-neutral field that caps the +context window an agent works within, set per agent config or per experiment variant. + +## Problem and evidence + +A benchmark of Claude Code on Sonnet 5 (Bedrock, 1M-token window) let runs grow to +250k-480k tokens of context. Above about 150k tokens each extended-thinking step got +about 10x slower, so most runs hit the 2-hour task limit. The operator wants to measure +pass rate and cost at several context caps, for example one variant at 200k and one at +400k, on the same tasks. + +Before this change no harness-neutral way existed: + +- **claude-code**: `claude_settings` (`--settings`) could carry `autoCompactWindow`. That + works for one harness only, and when `claude_settings` is a file path a variant cannot + merge one key into it. +- **codex**: `CodexAgentConfig` has no pass-through. `CodexAgent._build_thread_options` + builds the thread `config` dict itself, so no value reached Codex. +- **antigravity, opencode, pi, delegate**: no knob is wired. + +## How each harness controls its window + +### claude-code + +Verified in the CLI the SDK spawns. `claude-agent-sdk` runs its **bundled** CLI before it +looks for `claude` on `PATH` (`SubprocessCLITransport._find_cli`), so the CLI that runs is +the one the SDK wheel bundles, not the one the agent image installs with npm: + +| coder_eval | claude-agent-sdk (uv.lock) | bundled CLI | image npm CLI | +|---|---|---|---| +| 0.12.1 | 0.2.124 | 2.1.216 | 2.1.177 | +| 0.12.12 | 0.2.159 | 2.1.281 | 2.1.281 | + +`environment_info.claude_code_cli` records `claude -v` from `PATH`, so a 0.12.1 run +records 2.1.177 while 2.1.216 ran. That is a separate defect and is not fixed here. + +Three inputs set the auto-compact window. Both bundled CLIs (2.1.216 and 2.1.281) contain +the same resolution function. The highest precedence comes first: + +1. `CLAUDE_CODE_AUTO_COMPACT_WINDOW` environment variable: a plain token count. It + outranks the flag and every settings scope. +2. `--autocompact `: added in 2.1.221 (CLI reference). It is absent from + 2.1.216 as a parsed option. +3. `autoCompactWindow` settings key. The settings schema is + `int().min(100000).max(1000000).optional().catch(undefined)`, so an out-of-range + value is **silently dropped**. + +The effective window is `min(model context window, configured value)`. Auto-compaction +starts when the context approaches that window. + +### codex + +Verified in the `openai-codex-cli-bin` 0.156.1 binary (the pin) and its source at tag +`rust-v0.156.1`: + +- `model_context_window` ("Size of the context window for the model, in tokens"): + `with_config_overrides` sets the model's `context_window` to + `min(value, max_context_window)`. Three things follow from it. The default + auto-compact limit becomes 90% of it. The hard cap, which forces compaction, becomes + `context_window * effective_context_window_percent / 100`. The remaining-window figure + the model is shown also changes. +- `model_auto_compact_token_limit` ("Token usage threshold triggering auto-compaction"): + the trigger only, clamped to 90% of the resolved window. It does not change the window. +- `model_auto_compact_token_limit_scope` (`total` by default, or `body_after_prefix`) and + `model_post_turn_compact_threshold_percent` (turn-end compaction, off by default) tune + when compaction starts. They are not caps. + +The thread `config` dict `thread_start` takes is the same override surface that +coder_eval already uses for `enabled_tools` and `model_providers`. + +### antigravity, opencode, pi, delegate + +None of them has a context-window knob wired in coder_eval. They are out of scope, and +they reject the field. + +## Options considered + +**Per-harness pass-through** (a Codex `config` pass-through next to `claude_settings`). +It is flexible, but a variant would have to name a different key per harness, and the +recorded value would mean something different on each. It also opens a free-form +pass-through on Codex with no denylist. Rejected. + +**A field under `run_limits`.** Run limits are caps that the orchestrator or the agent +enforces on a run (turns, time, tokens, USD), and the parity table lists them. A context +window is not a spend cap. It is a harness setting that changes what the agent does, and +it depends on the agent type, which `run_limits` does not know. The guide also says that +`agent:` holds no run-time caps. Rejected. + +**A field on the agent config.** One name, accepted on a type-less config so that +experiment defaults and variants can set it, and validated against the concrete type +when the config resolves. **Chosen.** + +Naming: `context_window`, in tokens. It names the quantity the operator caps, matches +Codex's `model_context_window`, and avoids "auto-compact", which describes a mechanism of +one harness. `max_context_tokens` was considered, but it reads as a per-request input +limit, which neither harness enforces. + +## Recommended design + +`BaseAgentConfig.context_window: int | None` (default `None`, `gt=0`). + +- Each config class declares `_context_window_range: ClassVar[tuple[int, int | None] | + None]`. `None` (the base default) means that the harness cannot apply the field. + `ClaudeCodeAgentConfig` declares `(100_000, 1_000_000)` and `CodexAgentConfig` declares + `(1, None)`. +- `check_context_window_supported` (model validator) rejects an unsupported type or an + out-of-range value as soon as the type is known. A plugin agent inherits `None`, so it + also rejects the field until it opts in. +- The field uses the default `replace` merge strategy. It is set at any of the five + layers, and with `-D agent.context_window=N`. + +Per-harness mapping: + +- **claude-code**: `CLAUDE_CODE_AUTO_COMPACT_WINDOW=` in the SDK `env`. The + environment variable was chosen over the settings key and the flag for four reasons. + It has the highest precedence, so host, project and `claude_settings` values cannot + override the cap. It works whether `claude_settings` is a dict or a file path. Both + bundled CLIs read it, which the flag does not. `extra_args` stays framework-owned. + `claude_settings.autoCompactWindow` together with `context_window` is a load error, + because the environment variable would silently override it. +- **codex**: `model_context_window: ` in the thread `config`. No trigger key is set, + so Codex compacts at its default 90% of the window. + +## Validation + +- `gt=0` on every config, so a type-less config rejects nonsense as well. +- claude-code: 100000-1000000, the CLI's documented range. Outside that range the CLI + drops a settings value silently, so coder_eval enforces the range itself. +- codex: any positive integer. Codex clamps the value to the model's catalog maximum. +- every other type: rejected at load, naming the supported types. A variant that + switches `type` to an unsupported harness fails when the experiment resolves. + +## Recording + +The resolved agent config is persisted as `agent_config` on every task row (`run.json` +and `task.json`). `agent.context_window` therefore appears per task, and config lineage +records which layer set it (for example, the variant). The experiment report groups by +variant, so a cap per variant groups without new report code. On claude-code the +`sdk_options` dump also shows the environment variable in `env`. + +## Open questions + +- **A silent clamp to the model's window.** Neither harness reports that the model's + window lowered the cap. A cap above the model's window runs at the model's window. + Recording the effective window would need a per-model catalog in coder_eval. +- **Trigger points differ.** Codex compacts at 90% of the window. Claude Code compacts + when the context approaches the window, minus an internal buffer. Equal caps therefore + do not compact at exactly the same token count. A separate harness-neutral trigger + field is possible, but it is not added until a benchmark needs it. +- **A host `CLAUDE_CODE_AUTO_COMPACT_WINDOW` leaks into uncapped runs.** The SDK starts + the CLI with `os.environ` underneath `options.env`. When the field is unset, a host + value still applies. Clearing it would change behaviour for current users. It is + recorded in this note but not changed. +- **The recorded CLI version is wrong.** See the version table above: + `environment_info.claude_code_cli` should record the bundled CLI version. +- **Other harnesses.** OpenCode (`limit.context` in provider config) and Pi may have + equivalents. Each one needs its own verification before it opts in. diff --git a/docs/AB_EXPERIMENTS.md b/docs/AB_EXPERIMENTS.md index afe35d66b..2a7ee245a 100644 --- a/docs/AB_EXPERIMENTS.md +++ b/docs/AB_EXPERIMENTS.md @@ -23,6 +23,7 @@ to the orchestrator are involved. - [What a Variant Can Override](#what-a-variant-can-override) - [Recipe: A/B a Skill](#recipe-ab-a-skill) - [Recipe: A/B a Model](#recipe-ab-a-model) +- [Recipe: A/B a Context Window](#recipe-ab-a-context-window) - [Recipe: A/B a Prompt](#recipe-ab-a-prompt) - [Recipe: Smoke vs. e2e Flavors (Early Stop)](#recipe-smoke-vs-e2e-flavors-early-stop) - [Replicates (Statistical Power)](#replicates-statistical-power) @@ -222,6 +223,29 @@ variants: agent: { model: claude-opus-5 } ``` +## Recipe: A/B a Context Window + +`agent.context_window` caps the context the agent works within, in tokens: the +harness compacts the conversation before it outgrows the cap. Use it to measure +pass rate and cost against the cap on a model with a large window. Each task row +records the cap in `agent_config.context_window`. + +```yaml +experiment_id: context-window +description: "The same model with a 200k and a 400k context cap" + +variants: + - variant_id: cap-200k + agent: { context_window: 200000 } + - variant_id: cap-400k + agent: { context_window: 400000 } +``` + +Only `claude-code` and `codex` can apply the cap. On every other agent type the +experiment fails at load. See +[Run-Limit Parity](agents/HARNESS_PARITY.md#the-context-window-cap-per-harness) for what +the cap does on each harness. + ## Recipe: A/B a Prompt Use `prompt_mutations` (transform the task prompt) or `initial_prompt` / diff --git a/docs/TASK_DEFINITION_GUIDE.md b/docs/TASK_DEFINITION_GUIDE.md index 373210d52..1c98b389a 100644 --- a/docs/TASK_DEFINITION_GUIDE.md +++ b/docs/TASK_DEFINITION_GUIDE.md @@ -170,6 +170,7 @@ agent: - "Write" - "Bash" model: "claude-sonnet-5" # Optional: specific model + context_window: 200000 # Optional: cap the context window, in tokens (claude-code, codex) sdk_options: # Optional: agent SDK pass-through (keys depend on `type`) effort: high # claude-code: any non-framework-managed ClaudeAgentOptions field ``` @@ -188,6 +189,13 @@ validates its own keys at YAML load: Deep-merged across the 5-layer config chain. Override via CLI with the repeatable `-D agent.sdk_options.KEY=VALUE`. +**`context_window`** caps the context window the agent works within, in tokens: the +harness compacts the conversation before it outgrows the cap. Unset, the harness uses +the model's full window. `claude-code` accepts 100000-1000000 and `codex` any positive +value; every other agent type rejects the field at load. The cap never raises the +model's own window. Override via CLI with `-D agent.context_window=400000`. What each +harness does with it: [Run-Limit Parity](agents/HARNESS_PARITY.md#the-context-window-cap-per-harness). + **Permission Modes:** - `default` — Default permission handling - `acceptEdits` — Auto-accept file edits (recommended for evaluations) diff --git a/docs/agents/CLAUDE_CODE.md b/docs/agents/CLAUDE_CODE.md index 84e2146c8..8e7d535f2 100644 --- a/docs/agents/CLAUDE_CODE.md +++ b/docs/agents/CLAUDE_CODE.md @@ -104,6 +104,7 @@ agent: | `system_prompt_file` | `str \| null` | Path (relative to the task YAML) loaded into `system_prompt` at resolution. Works with either `system_prompt_mode`. | | `setting_sources` | `list["user"\|"project"\|"local"] \| null` | Which host setting sources the SDK reads. Default resolves to `["project"]`. See [Sandbox isolation](#sandbox-isolation). | | `claude_settings` | `str \| dict \| null` | Passed to the SDK `--settings`. A dict is JSON-serialized; a str is a settings file path. Use `permissions.deny` to block tools/paths. | +| `context_window` | `int \| null` (100000-1000000) | Auto-compact window, in tokens, sent as `CLAUDE_CODE_AUTO_COMPACT_WINDOW`. It outranks `--autocompact` and every settings scope, and is capped to the model's window. Setting `claude_settings.autoCompactWindow` too is a load error. See [Harness parity](HARNESS_PARITY.md#the-context-window-cap-per-harness). | | `sdk_options` | `dict` (default `{}`) | Pass-through for `ClaudeAgentOptions` fields Coder Eval doesn't own (e.g. `effort`). Validated at load — an unknown or framework-owned key is a hard error. | | `ignore_patterns` | `list[str] \| null` | Gitignore-style overrides for the workspace copy used by judge sub-agents (supports `!` negation). | diff --git a/docs/agents/CODEX.md b/docs/agents/CODEX.md index 29a985170..12b8e33e1 100644 --- a/docs/agents/CODEX.md +++ b/docs/agents/CODEX.md @@ -192,6 +192,13 @@ The agent maps `permission_mode` to the Codex SDK's `Sandbox`. The approval mode `allowed_tools` / `disallowed_tools` are normalized (`Bash` → `shell`, `Write`/`Edit` → `apply_patch`, etc.) and passed as `enabled_tools` / `disabled_tools` in the thread `config`. **Note:** the Codex SDK does not currently enforce `disabled_tools`; do not rely on it as a security boundary (the agent logs a warning when it is set). +### Context Window + +`agent.context_window` is passed as `model_context_window` in the thread `config`. Codex +uses it as the model's context window, capped to the model's catalog maximum, and +auto-compacts at 90% of it. See +[Harness parity](HARNESS_PARITY.md#the-context-window-cap-per-harness). + ### Skills Discovery The agent sets up SKILL.md files (Agent Skills open standard) in `.agents/skills/` directory: diff --git a/docs/agents/HARNESS_PARITY.md b/docs/agents/HARNESS_PARITY.md index 463655e9b..a3041e7bd 100644 --- a/docs/agents/HARNESS_PARITY.md +++ b/docs/agents/HARNESS_PARITY.md @@ -637,6 +637,30 @@ A timeout is a *failure* (partial turn captured, error status); the turn cap is *clean stop*. Conflating them is the mistake this page exists to prevent: a task whose cap fires should not look like a task whose harness hung. +## The context window cap per harness + +`agent.context_window` is not a run limit, but the same promise applies: one value +must mean the same cap on every harness that accepts it. It is the context window, in +tokens, that the harness compacts the conversation within. A harness that cannot apply +it rejects it when the config loads. It is never ignored, because a variant that runs +uncapped under a "capped" label is a wrong measurement. + +| | claude-code | codex | antigravity | opencode | pi | delegate | +|---|---|---|---|---|---|---| +| mechanism | `CLAUDE_CODE_AUTO_COMPACT_WINDOW` in the CLI environment | thread config `model_context_window` | rejected | rejected | rejected | rejected | +| accepted range | 100000-1000000 | any positive integer | | | | | +| when compaction starts | when the context approaches the window | at 90% of the window (Codex's default auto-compact limit) | | | | | +| above the model's own window | capped to the model's window by the CLI | capped to the model's catalog maximum by Codex | | | | | + +- **claude-code**: the environment variable outranks `--autocompact` and every settings + scope, so a host or project `autoCompactWindow` cannot change the cap. Setting + `claude_settings.autoCompactWindow` together with `context_window` fails at load. +- **codex**: the value replaces the model's context window, so the hard-cap + compaction at the usable window (a per-model share of it, 95% for Codex's fallback + model metadata) also moves. +- Neither harness reports when the model's own window has lowered the cap. A cap + above the model's window runs at the model's window. + ## `agent.plugins[].path` accepts different depths per harness Not a run limit, but the same promise: one task file, three harnesses, same meaning. diff --git a/src/coder_eval/agents/claude_code_agent.py b/src/coder_eval/agents/claude_code_agent.py index ecde1c383..1a93d11b4 100644 --- a/src/coder_eval/agents/claude_code_agent.py +++ b/src/coder_eval/agents/claude_code_agent.py @@ -180,6 +180,8 @@ def _is_sdk_result_message(message: Any) -> bool: _JSON_START_SEARCH_LIMIT = 200 +AUTO_COMPACT_WINDOW_ENV = "CLAUDE_CODE_AUTO_COMPACT_WINDOW" + class _ClaudeTurnState: """Per-turn mutable scratch state for one ``ClaudeCodeAgent.communicate`` call. @@ -1184,6 +1186,14 @@ def _build_claude_query( cost_log_tags=cost_log_tags, ) effective_model = self._resolve_effective_model(self.config.model, env, route_model) + if self.config.context_window is not None: + # The env var outranks --autocompact and every settings scope, and is + # read by every CLI the pinned SDKs bundle. + env[AUTO_COMPACT_WINDOW_ENV] = str(self.config.context_window) + elif os.environ.get(AUTO_COMPACT_WINDOW_ENV): + # The SDK spawns the CLI with {**os.environ, **options.env}, so a host-set + # window would cap a variant recorded as uncapped. Empty is falsy there. + env[AUTO_COMPACT_WINDOW_ENV] = "" disallowed_tools = list(self.config.disallowed_tools or []) # Do not allow ToolSearch. This is required to keep Bedrock backend in sync with the other backends. diff --git a/src/coder_eval/agents/codex_agent.py b/src/coder_eval/agents/codex_agent.py index bb3746a1b..2430eda3b 100644 --- a/src/coder_eval/agents/codex_agent.py +++ b/src/coder_eval/agents/codex_agent.py @@ -1500,6 +1500,10 @@ def _build_thread_options(self) -> dict[str, Any]: + f"api_version={'set' if api_version else 'unset'})" ) + if self.config.context_window is not None: + # Codex then auto-compacts at 90% of this window. + tool_config["model_context_window"] = self.config.context_window + if tool_config: options["config"] = tool_config diff --git a/src/coder_eval/models/agent_config.py b/src/coder_eval/models/agent_config.py index 27ce59742..1876d1df0 100644 --- a/src/coder_eval/models/agent_config.py +++ b/src/coder_eval/models/agent_config.py @@ -122,6 +122,9 @@ class BaseAgentConfig(BaseModel): # Cross-field merge exclusion: setting either prompt field at any layer clears # the sibling. ClassVar -> not a model field. _merge_exclusive_groups: ClassVar[tuple[tuple[str, ...], ...]] = (("system_prompt", "system_prompt_file"),) + # Inclusive (min, max) tokens the harness can cap its context window to; None + # means the harness has no such knob and ``context_window`` is rejected. + _context_window_range: ClassVar[tuple[int, int | None] | None] = None type: str | None = Field( default=None, @@ -162,6 +165,16 @@ class BaseAgentConfig(BaseModel): "Mutually exclusive with system_prompt." ), ) + context_window: int | None = Field( + default=None, + gt=0, + description=( + "Cap, in tokens, on the context window the agent works within: the harness compacts " + "the conversation before it outgrows this size. None leaves the harness default (the " + "model's full window). Supported by claude-code (100000-1000000) and codex; any other " + "agent type rejects it at load. Never raises the model's own maximum." + ), + ) # Customizable ignore patterns for file tracking ignore_patterns: list[str] = MergeField( @@ -203,12 +216,43 @@ def check_prompt_exclusivity(self) -> Self: raise ValueError("Only one of 'system_prompt' or 'system_prompt_file' can be provided, not both") return self + @model_validator(mode="after") + def check_context_window_supported(self) -> Self: + """Reject a ``context_window`` the resolved agent type cannot apply. + + A type-less config defers the check to the concrete config it resolves to. + An unsupported or out-of-range value fails here, at load, because a harness + that drops it would run the variant uncapped and still label it capped. + + Rationale: .claude/notes/context-window.md § Recommended design + """ + if self.context_window is None or self.type is None: + return self + bounds = type(self)._context_window_range + if bounds is None: + raise ValueError( + f"context_window is not supported by agent type {self.type!s}; " + + f"supported agent types: {', '.join(_kinds_supporting_context_window())}" + ) + low, high = bounds + if self.context_window < low or (high is not None and self.context_window > high): + allowed = f"{low}-{high}" if high is not None else f">= {low}" + raise ValueError( + f"context_window {self.context_window} is out of range for agent type {self.type!s} " + + f"(allowed: {allowed} tokens)" + ) + return self + class ClaudeCodeAgentConfig(BaseAgentConfig): """Claude Code agent configuration.""" type: Literal[AgentKind.CLAUDE_CODE] # type: ignore[assignment] + # The CLI's own accepted auto-compact window range; it silently drops a + # settings value outside it, so the bound is enforced here instead. + _context_window_range: ClassVar[tuple[int, int | None] | None] = (100_000, 1_000_000) + system_prompt_mode: SystemPromptMode = Field( default="append", description=( @@ -285,12 +329,33 @@ def check_replace_mode_has_prompt(self) -> Self: ) return self + @model_validator(mode="after") + def check_single_auto_compact_source(self) -> Self: + """Reject ``context_window`` alongside ``claude_settings.autoCompactWindow``. + + ``context_window`` reaches the CLI as ``CLAUDE_CODE_AUTO_COMPACT_WINDOW``, which + outranks the settings key, so the settings value would be silently ignored. + """ + if ( + self.context_window is not None + and isinstance(self.claude_settings, dict) + and "autoCompactWindow" in self.claude_settings + ): + raise ValueError( + "context_window and claude_settings.autoCompactWindow both set the auto-compact window; " + + "set only context_window" + ) + return self + class CodexAgentConfig(BaseAgentConfig): """Codex agent configuration.""" type: Literal[AgentKind.CODEX] # type: ignore[assignment] + # Codex clamps the value to the model's catalog maximum itself. + _context_window_range: ClassVar[tuple[int, int | None] | None] = (1, None) + # Mirrors google.antigravity.types.ThinkingLevel as a plain Literal so this module # imports without the optional SDK -- base installs must load every config class. @@ -546,6 +611,17 @@ def parse_agent_config(**kwargs: Any) -> BaseAgentConfig: return registration.config_class.model_validate(kwargs) +def _kinds_supporting_context_window() -> list[str]: + """Registered agent kinds whose config class can apply ``context_window``.""" + from coder_eval.agents.registry import AgentRegistry + + return [ + kind + for kind in sorted(AgentRegistry.list_kinds()) + if (reg := AgentRegistry.get(kind)) is not None and reg.config_class._context_window_range is not None + ] + + def _coerce_agent_config(value: Any) -> Any: """Coerce a raw agent-config dict to its registered subclass via the registry. diff --git a/tests/test_agent_context_window.py b/tests/test_agent_context_window.py new file mode 100644 index 000000000..108510e77 --- /dev/null +++ b/tests/test_agent_context_window.py @@ -0,0 +1,121 @@ +"""``agent.context_window``: one harness-neutral cap, applied per harness or rejected at load.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest +from pydantic import ValidationError + +from coder_eval.agents.claude_code_agent import AUTO_COMPACT_WINDOW_ENV, ClaudeCodeAgent +from coder_eval.models import AgentKind, BaseAgentConfig, TaskDefinition, parse_agent_config +from coder_eval.orchestration.config_merge import Layer, resolve_root, validate_paths +from coder_eval.orchestration.overrides import OverrideError, apply_overrides + + +def _claude_env(**config_kwargs) -> dict[str, str]: + agent = ClaudeCodeAgent(parse_agent_config(type=AgentKind.CLAUDE_CODE, **config_kwargs)) + agent.working_directory = Path(".") + options, _transport, _model = agent._build_claude_query("hi", None, None, lambda _line: None) + return options.env + + +def _task(agent: dict) -> TaskDefinition: + return TaskDefinition.model_validate( + {"task_id": "cw", "description": "d", "initial_prompt": "p", "agent": agent, "success_criteria": []} + ) + + +class TestClaudeCode: + def test_a_capped_variant_sends_the_window_to_the_cli(self): + assert _claude_env(context_window=200_000)[AUTO_COMPACT_WINDOW_ENV] == "200000" + + def test_an_uncapped_variant_leaves_the_cli_default(self): + assert AUTO_COMPACT_WINDOW_ENV not in _claude_env() + + def test_an_uncapped_variant_ignores_a_window_set_on_the_host(self, monkeypatch): + monkeypatch.setenv(AUTO_COMPACT_WINDOW_ENV, "150000") + assert _claude_env()[AUTO_COMPACT_WINDOW_ENV] == "" + assert _claude_env(context_window=300_000)[AUTO_COMPACT_WINDOW_ENV] == "300000" + + @pytest.mark.parametrize("window", [99_999, 1_000_001]) + def test_a_window_outside_the_cli_range_fails_at_load(self, window): + with pytest.raises(ValidationError, match="out of range for agent type claude-code"): + parse_agent_config(type=AgentKind.CLAUDE_CODE, context_window=window) + + def test_a_second_auto_compact_source_fails_at_load(self): + with pytest.raises(ValidationError, match=r"claude_settings.autoCompactWindow"): + parse_agent_config( + type=AgentKind.CLAUDE_CODE, + context_window=200_000, + claude_settings={"autoCompactWindow": 400_000}, + ) + + +class TestCodex: + def test_a_capped_variant_sets_the_model_context_window(self): + pytest.importorskip("openai_codex") + from coder_eval.agents.codex_agent import CodexAgent + + agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX, context_window=400_000)) + assert agent._build_thread_options()["config"]["model_context_window"] == 400_000 + + def test_an_uncapped_variant_leaves_the_codex_default(self): + pytest.importorskip("openai_codex") + from coder_eval.agents.codex_agent import CodexAgent + + options = CodexAgent(parse_agent_config(type=AgentKind.CODEX))._build_thread_options() + assert "model_context_window" not in options.get("config", {}) + + +class TestUnsupportedHarnesses: + @pytest.mark.parametrize( + "kind", [AgentKind.ANTIGRAVITY, AgentKind.OPENCODE, AgentKind.PI, AgentKind.DELEGATE, AgentKind.NONE] + ) + def test_a_harness_without_the_knob_fails_at_load(self, kind): + with pytest.raises(ValidationError, match="supported agent types: claude-code, codex"): + parse_agent_config(type=kind, context_window=200_000) + + def test_a_non_positive_window_fails_even_before_the_type_is_known(self): + with pytest.raises(ValidationError): + parse_agent_config(context_window=0) + + +class TestConfigLayers: + def test_a_type_less_task_defers_the_check_to_the_resolved_type(self): + assert isinstance(parse_agent_config(context_window=50_000), BaseAgentConfig) + + def test_a_variant_caps_the_task_it_runs_and_records_it(self): + lineage: dict = {} + resolved = resolve_root( + "agent", + [ + Layer(source="task", patch={"type": "claude-code"}), + Layer(source="variant", patch={"context_window": 200_000}, detail="variant cap-200k"), + ], + lineage=lineage, + ) + assert resolved is not None + assert resolved.model_dump()["context_window"] == 200_000 + assert lineage["agent.context_window"].source == "variant" + + def test_a_variant_switching_to_an_unsupported_type_fails_at_load(self): + with pytest.raises(ValidationError, match="not supported by agent type opencode"): + resolve_root( + "agent", + [ + Layer(source="experiment-defaults", patch={"context_window": 200_000}), + Layer(source="variant", patch={"type": "opencode"}), + ], + ) + + def test_a_cli_override_sets_the_cap(self): + validate_paths(["agent.context_window"]) + task = _task({"type": "codex"}) + apply_overrides(task, {"agent.context_window": 400_000}) + assert task.agent is not None and task.agent.context_window == 400_000 + + def test_a_cli_override_out_of_range_reads_as_a_clean_error(self): + task = _task({"type": "claude-code"}) + with pytest.raises(OverrideError, match=r"-D agent: .*out of range"): + apply_overrides(task, {"agent.context_window": 50_000}) From d077ad89a5df48e31327f37dfe16717a3e1a3cab Mon Sep 17 00:00:00 2001 From: Mihai Gheorghe Date: Thu, 8 Oct 2026 07:40:01 +0300 Subject: [PATCH 2/2] feat(agent): cap Pi's context window with agent.context_window Pi 0.87.1 can lower a model's window with providers..modelOverrides..contextWindow in models.json, and then compacts once the context passes the window minus compaction.reserveTokens, the same effect as Codex's model_context_window. Pi reads models.json only from its agent dir, so a capped agent runs from a temp dir that links every entry of the host's agent dir (auth, settings, extensions) and holds the host's models.json plus the override, pointed at by PI_CODING_AGENT_DIR and removed in stop(). Links keep credentials in one place and the capped variant's setup identical to the uncapped one's. Pi silently ignores an override for an unknown model id, so start() runs `pi --list-models ` against the mirror and fails unless that exact provider/model row shows the capped window. The config needs model as provider/model and a window of at least 32768 (twice the default reserve). OpenCode, Antigravity and delegate still reject the field; the note records why. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01RarvxSsr4K2KkBisTbXxQY --- .claude/notes/context-window.md | 48 +++++++-- docs/TASK_DEFINITION_GUIDE.md | 2 +- docs/agents/HARNESS_PARITY.md | 14 ++- docs/agents/PI.md | 14 +++ src/coder_eval/agents/pi_agent.py | 139 ++++++++++++++++++++++++++ src/coder_eval/models/agent_config.py | 21 +++- tests/test_agent_context_window.py | 91 ++++++++++++++++- 7 files changed, 311 insertions(+), 18 deletions(-) diff --git a/.claude/notes/context-window.md b/.claude/notes/context-window.md index 8086be927..89f4706da 100644 --- a/.claude/notes/context-window.md +++ b/.claude/notes/context-window.md @@ -18,7 +18,8 @@ Before this change no harness-neutral way existed: merge one key into it. - **codex**: `CodexAgentConfig` has no pass-through. `CodexAgent._build_thread_options` builds the thread `config` dict itself, so no value reached Codex. -- **antigravity, opencode, pi, delegate**: no knob is wired. +- **pi**: a window can be set per model in `models.json`, but coder_eval wrote none. +- **antigravity, opencode, delegate**: no knob is wired. ## How each harness controls its window @@ -70,10 +71,41 @@ Verified in the `openai-codex-cli-bin` 0.156.1 binary (the pin) and its source a The thread `config` dict `thread_start` takes is the same override surface that coder_eval already uses for `enabled_tools` and `model_providers`. -### antigravity, opencode, pi, delegate - -None of them has a context-window knob wired in coder_eval. They are out of scope, and -they reject the field. +### pi + +Pi 0.87.1 reads `models.json` from its agent dir (`PI_CODING_AGENT_DIR`, default +`~/.pi/agent`). `providers..modelOverrides..contextWindow` replaces a +built-in model's window as the topmost config layer (`dist/core/provider-composer.js`, +`applyModelOverride`), and Pi compacts by summarising once the context passes +`contextWindow - compaction.reserveTokens` (16384 by default; +`dist/core/compaction/compaction.js`). That is the same effect as Codex's +`model_context_window`. + +Pi has no separate path for `models.json`, so a capped agent runs from a per-agent +temp dir that links every entry of the host's agent dir (auth, settings, extensions, +prompts) and holds the host's `models.json` plus the override. Links rather than +copies keep credentials in one place and keep the capped variant's setup identical to +the uncapped one's; where the OS refuses a link the entry is copied. The temp dir is +removed in `stop()`. + +Pi silently ignores an override for a model id it does not know, so `start()` runs +`pi --list-models ` against the mirror and fails unless that exact +provider/model row shows the capped window. A host `models.json` with comments (Pi +strips them) cannot be merged as JSON and fails `start()` the same way. + +### antigravity, opencode, delegate + +None of them has a context-window knob wired in coder_eval, and they reject the field. + +- **opencode**: `provider..models..limit.context` (and `limit.input`, which + models that declare it use instead) would work, but the schema also requires + `limit.output`, which coder_eval does not know per model, and the OpenCode CLI is not + version-pinned. +- **antigravity**: `CompactionConfig(token_threshold=N)` in `LocalAgentConfig` + (google-antigravity 0.1.20) moves the compaction trigger without lowering the + window; what the closed binary does at the threshold has not been observed. +- **delegate**: the conversation lives and is summarised in the UiPath backend, and + no client option controls it. ## Options considered @@ -129,6 +161,8 @@ Per-harness mapping: - claude-code: 100000-1000000, the CLI's documented range. Outside that range the CLI drops a settings value silently, so coder_eval enforces the range itself. - codex: any positive integer. Codex clamps the value to the model's catalog maximum. +- pi: at least 32768, twice Pi's default compaction reserve, and `model` must be in + `provider/model` form, since the override is per model. - every other type: rejected at load, naming the supported types. A variant that switches `type` to an unsupported harness fails when the experiment resolves. @@ -155,5 +189,5 @@ variant, so a cap per variant groups without new report code. On claude-code the recorded in this note but not changed. - **The recorded CLI version is wrong.** See the version table above: `environment_info.claude_code_cli` should record the bundled CLI version. -- **Other harnesses.** OpenCode (`limit.context` in provider config) and Pi may have - equivalents. Each one needs its own verification before it opts in. +- **Other harnesses.** OpenCode can opt in once its CLI is pinned and coder_eval can + supply `limit.output`; Antigravity once one run shows what `token_threshold` does. diff --git a/docs/TASK_DEFINITION_GUIDE.md b/docs/TASK_DEFINITION_GUIDE.md index 1c98b389a..c1f2871a0 100644 --- a/docs/TASK_DEFINITION_GUIDE.md +++ b/docs/TASK_DEFINITION_GUIDE.md @@ -170,7 +170,7 @@ agent: - "Write" - "Bash" model: "claude-sonnet-5" # Optional: specific model - context_window: 200000 # Optional: cap the context window, in tokens (claude-code, codex) + context_window: 200000 # Optional: cap the context window, in tokens (claude-code, codex, pi) sdk_options: # Optional: agent SDK pass-through (keys depend on `type`) effort: high # claude-code: any non-framework-managed ClaudeAgentOptions field ``` diff --git a/docs/agents/HARNESS_PARITY.md b/docs/agents/HARNESS_PARITY.md index a3041e7bd..a3f229caf 100644 --- a/docs/agents/HARNESS_PARITY.md +++ b/docs/agents/HARNESS_PARITY.md @@ -647,10 +647,10 @@ uncapped under a "capped" label is a wrong measurement. | | claude-code | codex | antigravity | opencode | pi | delegate | |---|---|---|---|---|---|---| -| mechanism | `CLAUDE_CODE_AUTO_COMPACT_WINDOW` in the CLI environment | thread config `model_context_window` | rejected | rejected | rejected | rejected | -| accepted range | 100000-1000000 | any positive integer | | | | | -| when compaction starts | when the context approaches the window | at 90% of the window (Codex's default auto-compact limit) | | | | | -| above the model's own window | capped to the model's window by the CLI | capped to the model's catalog maximum by Codex | | | | | +| mechanism | `CLAUDE_CODE_AUTO_COMPACT_WINDOW` in the CLI environment | thread config `model_context_window` | rejected | rejected | `modelOverrides..contextWindow` in a per-agent `models.json` | rejected | +| accepted range | 100000-1000000 | any positive integer | | | >= 32768, `model` as `provider/model` | | +| when compaction starts | when the context approaches the window | at 90% of the window (Codex's default auto-compact limit) | | | past the window minus `compaction.reserveTokens` (16384 by default) | | +| above the model's own window | capped to the model's window by the CLI | capped to the model's catalog maximum by Codex | | | replaces the model's window | | - **claude-code**: the environment variable outranks `--autocompact` and every settings scope, so a host or project `autoCompactWindow` cannot change the cap. Setting @@ -658,7 +658,11 @@ uncapped under a "capped" label is a wrong measurement. - **codex**: the value replaces the model's context window, so the hard-cap compaction at the usable window (a per-model share of it, 95% for Codex's fallback model metadata) also moves. -- Neither harness reports when the model's own window has lowered the cap. A cap +- **pi**: Pi reads `models.json` from its agent dir only, so a capped agent runs from a + temp dir that links the host's agent dir (auth, settings, extensions) and adds the + override. `start()` fails unless `pi --list-models` shows the capped window for that + exact model, because Pi silently ignores an override for an unknown model id. +- Neither claude-code nor codex reports when the model's own window has lowered the cap. A cap above the model's window runs at the model's window. ## `agent.plugins[].path` accepts different depths per harness diff --git a/docs/agents/PI.md b/docs/agents/PI.md index eb5151971..1e2b5af89 100644 --- a/docs/agents/PI.md +++ b/docs/agents/PI.md @@ -112,6 +112,20 @@ Pi's reasoning effort, forwarded as `--thinking`. Accepts the seven-value set `off` / `minimal` / `low` / `medium` / `high` / `xhigh` / `max` (a strict superset of the Antigravity `thinking_level`), defaulting to `medium`. +### `context_window` + +`agent.context_window` caps the model's context window, in tokens (at least 32768). +It needs `model` in `provider/model` form, because Pi applies it as +`providers..modelOverrides..contextWindow` in `models.json`. Pi then +compacts once the context passes the window minus `compaction.reserveTokens`. + +Pi reads `models.json` only from its agent dir, so a capped agent runs from a temp dir +that links the host's agent dir (auth, settings, extensions) and holds the host's +`models.json` plus the override. `start()` fails unless `pi --list-models` shows the +capped window for that exact model, since Pi silently ignores an override for a model +id it does not know. See +[Harness parity](HARNESS_PARITY.md#the-context-window-cap-per-harness). + ### Enforced config fields Pi forwards these config knobs to real CLI flags: diff --git a/src/coder_eval/agents/pi_agent.py b/src/coder_eval/agents/pi_agent.py index 01ef13bbc..e40c5168d 100644 --- a/src/coder_eval/agents/pi_agent.py +++ b/src/coder_eval/agents/pi_agent.py @@ -35,6 +35,7 @@ import time from collections.abc import Callable from datetime import datetime +from pathlib import Path from typing import Any, ClassVar, Literal, NoReturn from uuid import uuid4 @@ -79,6 +80,78 @@ logger = logging.getLogger(__name__) +# Pi reads models.json, auth.json and settings.json from one agent directory and has no +# separate models path, so a context_window runs Pi from a per-agent mirror of that +# directory whose models.json adds the cap as a modelOverrides entry. +PI_AGENT_DIR_ENV = "PI_CODING_AGENT_DIR" +_PI_MODELS_FILE = "models.json" +_PI_LIST_MODELS_TIMEOUT_SECONDS = 60 + + +def _host_pi_agent_dir() -> Path: + configured = os.environ.get(PI_AGENT_DIR_ENV) + return Path(configured).expanduser() if configured else Path.home() / ".pi" / "agent" + + +def _pi_token_count(count: int) -> str: + """The token count as ``pi --list-models`` prints it (200000 -> "200K", 1000000 -> "1M").""" + if count >= 1_000_000: + millions = count / 1_000_000 + return f"{int(millions)}M" if millions % 1 == 0 else f"{millions:.1f}M" + if count >= 1_000: + thousands = count / 1_000 + return f"{int(thousands)}K" if thousands % 1 == 0 else f"{thousands:.1f}K" + return str(count) + + +def _mirror_agent_dir(host: Path, target: Path) -> None: + """Link every entry of the host's agent dir into ``target``, except models.json. + + Links, not copies, so credentials are not duplicated and the capped variant runs + with exactly the uncapped one's settings, extensions and prompts. Where the OS + refuses a link (Windows without the privilege), the entry is copied. + """ + if not host.is_dir(): + return + for entry in host.iterdir(): + if entry.name == _PI_MODELS_FILE: + continue + link = target / entry.name + try: + link.symlink_to(entry, target_is_directory=entry.is_dir()) + except OSError: + if entry.is_dir(): + shutil.copytree(entry, link, symlinks=True) + else: + shutil.copy2(entry, link) + + +def _capped_models_config(host: Path, provider: str, model_id: str, window: int) -> dict[str, Any]: + """The host's models.json with ``providers..modelOverrides..contextWindow`` set.""" + path = host / _PI_MODELS_FILE + try: + config: Any = json.loads(path.read_text(encoding="utf-8-sig")) if path.is_file() else {} + except json.JSONDecodeError as exc: + raise RuntimeError( + f"pi: cannot add context_window to {path}: it is not plain JSON ({exc}); " + + "coder_eval does not merge into a models.json with comments" + ) from exc + providers = config.setdefault("providers", {}) if isinstance(config, dict) else None + provider_config = providers.setdefault(provider, {}) if isinstance(providers, dict) else None + overrides = provider_config.setdefault("modelOverrides", {}) if isinstance(provider_config, dict) else None + model_override = overrides.setdefault(model_id, {}) if isinstance(overrides, dict) else None + if not isinstance(model_override, dict): + raise RuntimeError(f"pi: cannot add context_window to {path}: unexpected shape under providers.{provider}") + model_override["contextWindow"] = window + return config + + +def _write_capped_agent_dir(host: Path, target: Path, provider: str, model_id: str, window: int) -> None: + _mirror_agent_dir(host, target) + config = _capped_models_config(host, provider, model_id, window) + (target / _PI_MODELS_FILE).write_text(json.dumps(config, indent=2), encoding="utf-8") + + # Grace period between SIGTERM and SIGKILL when tearing down the CLI subprocess. # Doubles as the post-EOF exit grace in _settle_turn when no turn deadline is set. # Re-declared at OpenCode's value rather than shared — see the notes. @@ -715,6 +788,8 @@ def __init__( # Rationale: .claude/notes/agents.md § Reaping the CLI harnesses self._session_id: str | None = None self._session_dir: str | None = None + # The per-agent mirror of Pi's agent dir, set only when context_window is. + self._agent_dir: str | None = None self._process: asyncio.subprocess.Process | None = None # Process-group ids of every invocation this agent spawned, swept on # kill()/kill_sync()/stop(). @@ -768,11 +843,65 @@ async def start( safe_task_id = re.sub(r"[^A-Za-z0-9._-]", "_", self.task_id) self._session_id = f"coder-eval-{safe_task_id}-{uuid4().hex[:8]}" self._session_dir = tempfile.mkdtemp(prefix="pi-session-") + self._cleanup_agent_dir() + if self.config.context_window is not None: + await self._prepare_capped_agent_dir(self.config.context_window) self._state = AgentState.WORKING + async def _prepare_capped_agent_dir(self, window: int) -> None: + """Run Pi from a mirror of its agent dir whose models.json caps the model's window. + + Pi silently ignores an override for a model id it does not know, so the cap is + confirmed with ``pi --list-models`` before the run, and a mismatch fails start(). + """ + assert self.config.model is not None # PiAgentConfig rejects a window without one + provider, model_id = self.config.model.strip("/").split("/", 1) + host = _host_pi_agent_dir() + self._agent_dir = tempfile.mkdtemp(prefix="pi-agent-") + try: + await asyncio.to_thread(_write_capped_agent_dir, host, Path(self._agent_dir), provider, model_id, window) + listed = await self._listed_context(provider, model_id) + except BaseException: + self._cleanup_agent_dir() + raise + expected = _pi_token_count(window) + if listed != expected: + self._cleanup_agent_dir() + seen = f"lists it with context {listed}" if listed is not None else "does not list it" + raise RuntimeError( + f"pi: context_window {window} did not apply to {provider}/{model_id}: `pi --list-models` {seen}, " + + f"expected {expected}; check the model id and that its provider has credentials" + ) + logger.info("pi: %s/%s capped to a %s context window", provider, model_id, expected) + + async def _listed_context(self, provider: str, model_id: str) -> str | None: + """The context column ``pi --list-models`` prints for exactly this model, or None.""" + proc = await asyncio.create_subprocess_exec( + "pi", + "--list-models", + model_id, + stdout=asyncio.subprocess.PIPE, + stderr=asyncio.subprocess.PIPE, + limit=STDOUT_LINE_LIMIT_BYTES, + cwd=self.working_directory, + env=self._build_env(), + ) + try: + stdout, _stderr = await asyncio.wait_for(proc.communicate(), timeout=_PI_LIST_MODELS_TIMEOUT_SECONDS) + except TimeoutError: + with contextlib.suppress(ProcessLookupError): + proc.kill() + return None + for line in stdout.decode("utf-8", errors="replace").splitlines()[1:]: + columns = line.split() + if len(columns) >= 3 and columns[0] == provider and columns[1] == model_id: + return columns[2] + return None + async def stop(self) -> None: await self.kill() self._cleanup_session_dir() + self._cleanup_agent_dir() self._mark_stopped() def _cleanup_session_dir(self) -> None: @@ -780,6 +909,12 @@ def _cleanup_session_dir(self) -> None: shutil.rmtree(self._session_dir, ignore_errors=True) self._session_dir = None + def _cleanup_agent_dir(self) -> None: + # rmtree removes the links, never what they point at in the host's agent dir. + if self._agent_dir is not None: + shutil.rmtree(self._agent_dir, ignore_errors=True) + self._agent_dir = None + async def kill(self) -> None: proc = self._process if proc is not None and proc.returncode is None: @@ -823,6 +958,8 @@ def get_environment_info(self) -> dict[str, Any]: } if self._session_id: info["pi_session_id"] = self._session_id + if self._agent_dir is not None and self.config.context_window is not None: + info["pi_context_window"] = self.config.context_window if self._skill_dirs: # Recorded per task so a run's report can confirm the skills under test # actually reached the agent. @@ -879,6 +1016,8 @@ def _build_env(self) -> dict[str, str]: env["PATH"] = os.pathsep.join([*self._env_path_prepend, env.get("PATH", "")]) if self._plugin_tools_dir and "PLUGIN_TOOLS_DIR" not in env: env["PLUGIN_TOOLS_DIR"] = self._plugin_tools_dir + if self._agent_dir is not None: + env[PI_AGENT_DIR_ENV] = self._agent_dir return env # --- the turn ---------------------------------------------------------- diff --git a/src/coder_eval/models/agent_config.py b/src/coder_eval/models/agent_config.py index 1876d1df0..ce05e4233 100644 --- a/src/coder_eval/models/agent_config.py +++ b/src/coder_eval/models/agent_config.py @@ -171,7 +171,8 @@ class BaseAgentConfig(BaseModel): description=( "Cap, in tokens, on the context window the agent works within: the harness compacts " "the conversation before it outgrows this size. None leaves the harness default (the " - "model's full window). Supported by claude-code (100000-1000000) and codex; any other " + "model's full window). Supported by claude-code (100000-1000000), codex and pi " + "(>= 32768, with model as provider/model); any other " "agent type rejects it at load. Never raises the model's own maximum." ), ) @@ -452,11 +453,29 @@ class PiAgentConfig(BaseAgentConfig): type: Literal[AgentKind.PI] # type: ignore[assignment] + # Pi compacts once the context passes contextWindow minus its 16384-token + # compaction.reserveTokens, so the window has to leave room above that reserve. + _context_window_range: ClassVar[tuple[int, int | None] | None] = (32_768, None) + thinking_level: PiThinkingLevel = Field( default="medium", description="Pi reasoning effort passed as --thinking (off/minimal/low/medium/high/xhigh/max).", ) + @model_validator(mode="after") + def check_context_window_names_a_model(self) -> Self: + """Require ``provider/model`` when ``context_window`` is set. + + Pi caps a window through a per-model ``modelOverrides`` entry in ``models.json``, + so the cap needs the provider and the model id it applies to. + """ + if self.context_window is not None and (self.model is None or "/" not in self.model.strip("/")): + raise ValueError( + "context_window on pi needs model in Pi's provider/model form " + + "(e.g. anthropic/claude-sonnet-4-5): Pi caps a window per model" + ) + return self + _DELEGATE_SDK_OPTION_FIELDS: frozenset[str] = frozenset({"effort"}) diff --git a/tests/test_agent_context_window.py b/tests/test_agent_context_window.py index 108510e77..8bf9ea0de 100644 --- a/tests/test_agent_context_window.py +++ b/tests/test_agent_context_window.py @@ -2,12 +2,15 @@ from __future__ import annotations +import json from pathlib import Path import pytest from pydantic import ValidationError from coder_eval.agents.claude_code_agent import AUTO_COMPACT_WINDOW_ENV, ClaudeCodeAgent +from coder_eval.agents.pi_agent import _PI_MODELS_FILE as _PI_MODELS +from coder_eval.agents.pi_agent import PI_AGENT_DIR_ENV, PiAgent, _pi_token_count from coder_eval.models import AgentKind, BaseAgentConfig, TaskDefinition, parse_agent_config from coder_eval.orchestration.config_merge import Layer, resolve_root, validate_paths from coder_eval.orchestration.overrides import OverrideError, apply_overrides @@ -68,12 +71,92 @@ def test_an_uncapped_variant_leaves_the_codex_default(self): assert "model_context_window" not in options.get("config", {}) -class TestUnsupportedHarnesses: - @pytest.mark.parametrize( - "kind", [AgentKind.ANTIGRAVITY, AgentKind.OPENCODE, AgentKind.PI, AgentKind.DELEGATE, AgentKind.NONE] +PI_MODEL = "anthropic/claude-sonnet-4-5" + + +def _pi_host(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + host = tmp_path / "host-agent" + (host / "extensions").mkdir(parents=True) + (host / "auth.json").write_text('{"anthropic": {"type": "api_key"}}', encoding="utf-8") + (host / "settings.json").write_text('{"compaction": {"reserveTokens": 16384}}', encoding="utf-8") + (host / _PI_MODELS).write_text( + json.dumps({"providers": {"anthropic": {"modelOverrides": {"claude-opus-4-5": {"maxTokens": 8000}}}}}), + encoding="utf-8", ) + monkeypatch.setenv(PI_AGENT_DIR_ENV, str(host)) + monkeypatch.setattr("shutil.which", lambda _name: "/usr/local/bin/pi") + return host + + +async def _started_pi(tmp_path: Path, listed: str | None, **config_kwargs) -> PiAgent: + agent = PiAgent(parse_agent_config(type=AgentKind.PI, **config_kwargs)) + + async def list_models(_provider: str, _model_id: str) -> str | None: + return listed + + agent._listed_context = list_models # type: ignore[method-assign] + await agent.start(str(tmp_path)) + return agent + + +class TestPi: + async def test_a_capped_variant_runs_pi_from_a_mirror_whose_models_json_caps_the_model(self, tmp_path, monkeypatch): + host = _pi_host(tmp_path, monkeypatch) + agent = await _started_pi(tmp_path, "200K", model=PI_MODEL, context_window=200_000) + + mirror = Path(agent._build_env()[PI_AGENT_DIR_ENV]) + assert mirror != host + models = json.loads((mirror / _PI_MODELS).read_text(encoding="utf-8"))["providers"]["anthropic"] + assert models["modelOverrides"]["claude-sonnet-4-5"] == {"contextWindow": 200_000} + assert models["modelOverrides"]["claude-opus-4-5"] == {"maxTokens": 8000} + assert (mirror / "auth.json").read_text(encoding="utf-8") == (host / "auth.json").read_text(encoding="utf-8") + assert (mirror / "settings.json").exists() and (mirror / "extensions").is_dir() + assert agent.get_environment_info()["pi_context_window"] == 200_000 + + await agent.stop() + assert not mirror.exists() + assert json.loads((host / _PI_MODELS).read_text(encoding="utf-8"))["providers"]["anthropic"][ + "modelOverrides" + ] == {"claude-opus-4-5": {"maxTokens": 8000}} + assert (host / "auth.json").exists() and (host / "extensions").is_dir() + + async def test_a_cap_pi_does_not_apply_fails_the_start(self, tmp_path, monkeypatch): + _pi_host(tmp_path, monkeypatch) + agent = PiAgent(parse_agent_config(type=AgentKind.PI, model=PI_MODEL, context_window=200_000)) + + async def unknown_model(_provider: str, _model_id: str) -> str | None: + return None + + agent._listed_context = unknown_model # type: ignore[method-assign] + with pytest.raises(RuntimeError, match="did not apply to anthropic/claude-sonnet-4-5"): + await agent.start(str(tmp_path)) + assert agent._agent_dir is None + + async def test_an_uncapped_variant_keeps_the_host_agent_dir(self, tmp_path, monkeypatch): + host = _pi_host(tmp_path, monkeypatch) + agent = await _started_pi(tmp_path, None, model=PI_MODEL) + assert agent._build_env()[PI_AGENT_DIR_ENV] == str(host) + assert "pi_context_window" not in agent.get_environment_info() + + def test_a_window_needs_the_provider_and_model_it_caps(self): + with pytest.raises(ValidationError, match="provider/model form"): + parse_agent_config(type=AgentKind.PI, context_window=200_000) + with pytest.raises(ValidationError, match="provider/model form"): + parse_agent_config(type=AgentKind.PI, model="claude-sonnet-4-5", context_window=200_000) + + def test_a_window_under_the_compaction_reserve_fails_at_load(self): + with pytest.raises(ValidationError, match="out of range for agent type pi"): + parse_agent_config(type=AgentKind.PI, model=PI_MODEL, context_window=16_000) + + @pytest.mark.parametrize(("tokens", "printed"), [(200_000, "200K"), (1_000_000, "1M"), (32_768, "32.8K")]) + def test_the_check_reads_counts_as_pi_prints_them(self, tokens, printed): + assert _pi_token_count(tokens) == printed + + +class TestUnsupportedHarnesses: + @pytest.mark.parametrize("kind", [AgentKind.ANTIGRAVITY, AgentKind.OPENCODE, AgentKind.DELEGATE, AgentKind.NONE]) def test_a_harness_without_the_knob_fails_at_load(self, kind): - with pytest.raises(ValidationError, match="supported agent types: claude-code, codex"): + with pytest.raises(ValidationError, match="supported agent types: claude-code, codex, pi"): parse_agent_config(type=kind, context_window=200_000) def test_a_non_positive_window_fails_even_before_the_type_is_known(self):