From 9a288f3cc1c569ef9d030b24f6e8e3ffe610dd55 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Sat, 26 Sep 2026 23:40:16 +0800 Subject: [PATCH 01/12] docs: record experiment raising max_turns to 100 Both remaining reward-0 tasks (caffe-cifar-10, mteb-leaderboard) passed with max_turns=100. mteb-leaderboard finished cleanly on turn 70; caffe-cifar-10 passed from artifacts on disk after hitting the deadline. The run removed the Harbor timeout overrides to stay leaderboard-compliant. --- .../results/tb21-exp1-turn100-20260926.json | 91 +++++++++++++++++++ docs/dev_notes/en/0.8.x.md | 66 ++++++++++++++ docs/dev_notes/zh-CN/0.8.x.md | 56 ++++++++++++ 3 files changed, 213 insertions(+) create mode 100644 benchmarks/harbor/results/tb21-exp1-turn100-20260926.json diff --git a/benchmarks/harbor/results/tb21-exp1-turn100-20260926.json b/benchmarks/harbor/results/tb21-exp1-turn100-20260926.json new file mode 100644 index 0000000..19a770b --- /dev/null +++ b/benchmarks/harbor/results/tb21-exp1-turn100-20260926.json @@ -0,0 +1,91 @@ +{ + "schema_version": 1, + "generated_at": "2026-09-26T15:39:43.242545+00:00", + "purpose": "Experiment 1: raise the headless max_turns from 50 to 100 for the two tasks that still scored reward 0 in tb21-v41flash-6task-20260925.", + "agent_ref": "41476d451a16f6455f806f2b038b3d798537da01", + "agent_version": "0.8.1.dev32+g41476d451", + "model": "openrouter/deepseek/deepseek-v4.1-flash", + "dataset": "terminal-bench/terminal-bench-2-1", + "dataset_ref": "sha256:7d7bdc1cbedad549fc1140404bd4dc45e5fd0ea7c4186773687d177ad3a0699a", + "selection": "The two tasks that still scored reward 0 in tb21-v41flash-6task-20260925 (caffe-cifar-10, mteb-leaderboard) with their pinned task refs.", + "configuration": { + "max_turns": 100, + "previous_max_turns": 50, + "max_tokens": 65536, + "project_default_max_tokens": 32768, + "agent_execution_limit_seconds": 3480, + "finalization_grace_seconds": 120, + "harbor_outer_limit_seconds": null, + "agent_setup_limit_seconds": null, + "timeout_compliance": "Leaderboard-compliant: no Harbor agent/verifier timeout override and no setup timeout override; the effective agent timeout is the task-native 3600s. The agent self-limits at 3480s inside it.", + "maximum_simultaneous_trials": 2, + "automatic_harbor_retries": 0, + "attempts_per_task": 1, + "verifier_limits": "Native task deadlines", + "dependency_constraints": [ + "anthropic==1.5.0", + "httpx==0.28.1", + "httpx2==2.12.0", + "pydantic==2.13.5" + ] + }, + "baseline": { + "job": "tb21-v41flash-6task-20260925", + "max_turns": 50, + "note": "Both re-run tasks scored reward 0 there by exhausting 50 turns (max_turns_exhausted)." + }, + "totals": { + "planned": 2, + "finished": 2, + "passed": 2, + "scored_zero": 0, + "unscored": 0, + "model_attempts": 139, + "input_tokens": 340804, + "cache_tokens": 4289280, + "output_tokens": 73799, + "cost_usd": 0.094087476 + }, + "trials": [ + { + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "reward": 1.0, + "outcome": "DiagnosticDeadlineExceeded", + "duration_seconds": 3600.0, + "model_replies": 69, + "tool_calls": 77, + "input_tokens": 181427, + "cache_tokens": 2203008, + "output_tokens": 41101, + "cost_usd": 0.039895559, + "deadline_reached": true, + "forced_kill": true, + "trajectory_source": "reconstructed", + "started_at": "2026-09-26T14:36:51.471Z" + }, + { + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "reward": 1.0, + "outcome": "completed", + "duration_seconds": 1666.3, + "model_replies": 70, + "tool_calls": 90, + "input_tokens": 159377, + "cache_tokens": 2086272, + "output_tokens": 32698, + "cost_usd": 0.054191917, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "started_at": "2026-09-26T14:36:37.295Z" + } + ], + "comparison_limits": "Single attempt per task; routing, sampling, cache and provider backend are uncontrolled, so this is a targeted diagnostic, not a controlled estimate of pass-rate improvement.", + "artifacts": { + "raw_trials": "jobs/tb21-exp1-turn100-20260926", + "record": "jobs/tb21-exp1-turn100-20260926-record", + "note": "Raw trials and record live in Git-ignored jobs/ and are not uploaded." + } +} diff --git a/docs/dev_notes/en/0.8.x.md b/docs/dev_notes/en/0.8.x.md index 6057f38..9fb01b2 100644 --- a/docs/dev_notes/en/0.8.x.md +++ b/docs/dev_notes/en/0.8.x.md @@ -1131,3 +1131,69 @@ Whether this becomes a separate command or an automatic post-benchmark stage is **Evidence boundary.** Routing, sampling, cache, and backend were not controlled while the model changed, and each task ran only once, so **0/6 → 4/6 cannot be attributed to the model difference**; it is only a signal to verify later. **Artifacts.** Raw trials are under `jobs/tb21-v41flash-6task-20260925/`, with the experiment record under `jobs/tb21-v41flash-6task-20260925-record/` (`manifest.json`, `summary.json`, `report.md`); these live in Git-ignored `jobs/` and were not uploaded. Version control stores the structured result [`tb21-v41flash-6task-20260925.json`](../../../benchmarks/harbor/results/tb21-v41flash-6task-20260925.json) and this summary. + +### Experiment 1: raising max_turns to 100 + +**Run on 2026-09-26.** In the previous run (`tb21-v41flash-6task-20260925`), both +`caffe-cifar-10` and `mteb-leaderboard` exhausted 50 turns +(`max_turns_exhausted`). The hard constraint Terminal-Bench / Harbor imposes on an +agent is **wall-clock time** (this task's native `[agent].timeout_sec = 3600`); it +says nothing about a turn limit — the 50 turns are nanoPyCodeAgent's own cost +guardrail (`adapter.py:18`, `agent.py:82`). So this run changes one thing only: +**`max_turns` goes from 50 to 100, everything else stays the same**. To keep the +run compliant, the Harbor-level `override_timeout_sec` and +`override_setup_timeout_sec` were **removed**, so the task-native limits apply; the +agent self-limits to 3480s inside the task's 3600s and reserves 120s to finalize. + +| Item | This run | +| --- | --- | +| Job | `tb21-exp1-turn100-20260926` | +| Agent source | branch `experiment/max-turns-and-time-budget`, commit `41476d451`; wheel `0.8.1.dev32+g41476d451` | +| Model | `openrouter/deepseek/deepseek-v4.1-flash` | +| Dataset | `terminal-bench/terminal-bench-2-1`, `sha256:7d7bdc1c…a0699a` | +| Tasks | `caffe-cifar-10`, `mteb-leaderboard`, pinned task refs | +| Generation / turns | at most 65536 tokens per reply; at most **100** replies per task | +| Time limits | task-native agent 3600s; the agent self-limits to 3480s + 120s finalization; **no Harbor timeout overridden** | +| Concurrency / retries | concurrency 2; one attempt per task; zero automatic Harbor retries | + +**Result: both tasks passed, but they failed differently.** + +| Task | Reward | Terminal state | Model replies | Duration | Cost (USD) | +| --- | ---: | --- | ---: | ---: | ---: | +| `mteb-leaderboard` | 1 | `completed` | 70 | 1666 s | 0.054191917 | +| `caffe-cifar-10` | 1 | `DiagnosticDeadlineExceeded` | 69 | 3600 s | 0.039895559 | + +- `mteb-leaderboard`: **passed for the first time**, and cleanly — it wrote + `/app/result.txt` on turn 70 and stopped on its own. That confirms the previous + failure was **not a capability problem but the self-imposed 50-turn cap**: it had + already computed the right answer by turn 39, was still re-verifying at turn 50, + and finally converged before turn 100. +- `caffe-cifar-10`: all six official tests passed, but the agent did not wind down + on its own — after `Iteration 500` finished training and it had tested the + trained model, it went on to build pycaffe (`make py`) and install + `python3-skimage` and other extras until it hit the 3480s self-limit and the + 3600s hard limit and was killed (`deadline_reached=true`, + `forced_kill=true`). **The reward came from artifacts already on disk at the + timeout**, not from the agent finishing; the trajectory could only be + reconstructed by the supervisor from the journal + (`trajectory_source=reconstructed`). + +**Conclusion.** These two tasks show two sides of the same judgment: + +1. Raising the turn cap to 100 did let `mteb-leaderboard` pass — **the self-imposed + turn cap was the binding constraint**. +2. But `caffe-cifar-10` shows that **raising the cap alone does not teach the agent + to stop**: it keeps working until it is killed externally. That is exactly what + the time budget, budget injection, and stop-on-deadline (Experiment 2) are meant + to address. + +**Evidence boundary.** One trial per task, with routing, sampling, cache, and +backend uncontrolled; this is a targeted signal, not a controlled estimate of +pass-rate improvement. + +**Artifacts.** Raw trials are under `jobs/tb21-exp1-turn100-20260926/`, with the +experiment record under `jobs/tb21-exp1-turn100-20260926-record/`; these live in +Git-ignored `jobs/` and were not uploaded. Version control stores the structured +result +[`tb21-exp1-turn100-20260926.json`](../../../benchmarks/harbor/results/tb21-exp1-turn100-20260926.json) +and this summary. diff --git a/docs/dev_notes/zh-CN/0.8.x.md b/docs/dev_notes/zh-CN/0.8.x.md index 7cfe945..f12927a 100644 --- a/docs/dev_notes/zh-CN/0.8.x.md +++ b/docs/dev_notes/zh-CN/0.8.x.md @@ -1543,3 +1543,59 @@ V4.1 Flash 后: `report.md`);这些位于 Git 忽略的 `jobs/`,未上传。版本管理保存结构化结果 [`tb21-v41flash-6task-20260925.json`](../../../benchmarks/harbor/results/tb21-v41flash-6task-20260925.json) 与本节汇总。 + +### 实验一:把 max_turns 提到 100 + +**2026-09-26 实跑。** 上一轮(`tb21-v41flash-6task-20260925`)里 +`caffe-cifar-10` 与 `mteb-leaderboard` 都是 50 轮耗尽(`max_turns_exhausted`)。 +Terminal-Bench / Harbor 对 agent 的硬约束是**墙钟时间**(本题原生 +`[agent].timeout_sec = 3600`),并不包含轮数上限;50 轮是 nanoPyCodeAgent 自己的 +成本护栏(`adapter.py:18`、`agent.py:82`)。因此本轮只改一件事:**把 `max_turns` +从 50 提到 100,其余设置保持不变**。为让这次运行合规,**移除了 Harbor 层的 +`override_timeout_sec` 与 `override_setup_timeout_sec`**,改用题目原生时限,agent +在题目 3600 秒内自行限时 3480 秒并预留 120 秒收尾。 + +| 项目 | 本轮配置 | +| --- | --- | +| Job | `tb21-exp1-turn100-20260926` | +| Agent 源码 | 分支 `experiment/max-turns-and-time-budget`,提交 `41476d451`;wheel `0.8.1.dev32+g41476d451` | +| Model | `openrouter/deepseek/deepseek-v4.1-flash` | +| Dataset | `terminal-bench/terminal-bench-2-1`,`sha256:7d7bdc1c…a0699a` | +| 题目 | `caffe-cifar-10`、`mteb-leaderboard`,沿用固定 task ref | +| 生成/轮数 | 每次回复最多 65536 tokens;每题最多 **100** 次完整回复 | +| 时限 | 题目原生 agent 3600 秒;agent 自限 3480 秒 + 120 秒收尾;**未覆盖任何 Harbor timeout** | +| 并发/重试 | 并发 2;每题 1 次尝试;Harbor 自动重试 0 | + +**结果:2 题全部通过,但两题的死法不同。** + +| 题目 | Reward | 终态 | 模型回复 | 时长 | 费用(USD) | +| --- | ---: | --- | ---: | ---: | ---: | +| `mteb-leaderboard` | 1 | `completed` | 70 | 1666 s | 0.054191917 | +| `caffe-cifar-10` | 1 | `DiagnosticDeadlineExceeded` | 69 | 3600 s | 0.039895559 | + +- `mteb-leaderboard`:**历史上第一次通过**,而且是干净的 `completed`——第 70 轮 + 写完 `/app/result.txt` 并主动结束。这说明上一轮的失败**不是能力问题,而是 50 轮 + 这个自设上限卡住了它**:它在第 39 轮就已经算出正确答案,50 轮时还在复核,100 轮 + 时终于收敛。 +- `caffe-cifar-10`:官方 6 项测试全通过,但 agent 没有自己收尾——它在 + `Iteration 500` 训练完成、并用训练好的模型跑完测试之后,又去构建 pycaffe + (`make py`)、装 `python3-skimage` 等额外工作,直到撞上 3480 秒自限与 3600 秒 + 硬上限被强杀(`deadline_reached=true`、`forced_kill=true`)。**奖励来自超时前 + 已落盘的产物**,不是 agent 主动结束;轨迹因此只能由 supervisor 从 journal 重建 + (`trajectory_source=reconstructed`)。 + +**结论。** 这两题印证了同一判断的两面: + +1. 把轮数上限提到 100 确实让 `mteb-leaderboard` 通过——**自设的轮数上限曾经是 + 绑定约束**。 +2. 但 `caffe-cifar-10` 表明,**光放宽轮数并不能让 agent 学会收尾**:它仍会一路 + 做到被外部强杀。这正是"时间预算 + 预算注入 + 到点停止"(实验二)要解决的问题。 + +**证据边界。** 每题只跑 1 次,路由、采样、缓存、后端都未固定;这只是一个针对性 +信号,不是通过率的受控估计。 + +**产物位置。** 原始 trial 在 `jobs/tb21-exp1-turn100-20260926/`,实验记录在 +`jobs/tb21-exp1-turn100-20260926-record/`;这些位于 Git 忽略的 `jobs/`,未上传。 +版本管理保存结构化结果 +[`tb21-exp1-turn100-20260926.json`](../../../benchmarks/harbor/results/tb21-exp1-turn100-20260926.json) +与本节汇总。 From eb77fd578a3b08582f3f6958858c0756b66d58dc Mon Sep 17 00:00:00 2001 From: minixalpha Date: Sat, 26 Sep 2026 23:41:03 +0800 Subject: [PATCH 02/12] feat: add a wall-clock time budget to headless runs Headless runs can now be given a total wall-clock budget through --time-budget-seconds (Harbor kwarg time_budget_seconds). The loop tells the model how much time remains each turn, escalates to a stop-and-write warning inside a reserved finalization window, and stops with a new time_budget_exhausted outcome before the harness timeout kills the process. The budget is recorded in the journal and ATIF trajectory. --- .../harbor/src/harbor_adapter/adapter.py | 1 + docs/changelogs/0.8.x.md | 6 + src/nanopycodeagent/agent.py | 78 ++++++++- src/nanopycodeagent/atif.py | 5 + src/nanopycodeagent/cli.py | 12 ++ src/nanopycodeagent/event_journal.py | 6 +- tests/test_time_budget.py | 151 ++++++++++++++++++ 7 files changed, 254 insertions(+), 5 deletions(-) create mode 100644 tests/test_time_budget.py diff --git a/benchmarks/harbor/src/harbor_adapter/adapter.py b/benchmarks/harbor/src/harbor_adapter/adapter.py index 805810f..621eebb 100644 --- a/benchmarks/harbor/src/harbor_adapter/adapter.py +++ b/benchmarks/harbor/src/harbor_adapter/adapter.py @@ -39,6 +39,7 @@ class NanoPyCodeAgent(BaseInstalledAgent): default=_DEFAULT_MAX_TURNS, ), CliFlag("max_tokens", cli="--max-tokens", type="int"), + CliFlag("time_budget_seconds", cli="--time-budget-seconds", type="int"), ] def __init__(self, *args, git_ref: str | None = None, **kwargs): diff --git a/docs/changelogs/0.8.x.md b/docs/changelogs/0.8.x.md index d3318ce..6007a9a 100644 --- a/docs/changelogs/0.8.x.md +++ b/docs/changelogs/0.8.x.md @@ -9,6 +9,12 @@ All notable changes in the **0.8.x** release series are documented here. `ANTHROPIC_MAX_TOKENS`, including the user settings file and Harbor adapter. Record the effective limit in startup output, Event Journals, and ATIF trajectories while retaining compatibility with older journals. +- Add a wall-clock time budget to headless runs through + `--time-budget-seconds` and the Harbor adapter `time_budget_seconds` kwarg. + Tell the model how much time remains each turn, escalate to a stop-and-write + warning inside a reserved finalization window, and stop with a + `time_budget_exhausted` outcome before the harness timeout can kill the + process. Record the budget in startup output, Event Journals, and ATIF. ### Fixed - Return correctable tool errors for missing or incorrectly typed arguments and diff --git a/src/nanopycodeagent/agent.py b/src/nanopycodeagent/agent.py index cd6d9e3..9f4f863 100644 --- a/src/nanopycodeagent/agent.py +++ b/src/nanopycodeagent/agent.py @@ -84,6 +84,40 @@ # refuses it. DEFAULT_MAX_TURNS = 50 +# A headless run may also be given a wall-clock budget. When it is, the loop +# tells the model how much time is left and stops before the harness's own +# timeout can kill the process with nothing written. The last stretch is +# reserved so the model still has room to write the task's output file. +_FINALIZATION_RESERVE_SECONDS = 180 +_FINALIZATION_RESERVE_FRACTION = 0.15 + + +def _format_duration(total_seconds: float) -> str: + total = max(0, int(total_seconds)) + return f"{total // 60}:{total % 60:02d}" + + +def _time_budget_note( + *, turn: int, elapsed: float, budget: float, remaining: float +) -> str: + """The wall-clock reminder appended to the system prompt for one turn.""" + if remaining <= max( + _FINALIZATION_RESERVE_FRACTION * budget, _FINALIZATION_RESERVE_SECONDS + ): + return ( + f"[time budget] Turn {turn}. Only {_format_duration(remaining)} of " + f"{_format_duration(budget)} left. Stop investigating now. If the " + "task names an output file, write your best answer to it, verify " + "it exists, then reply with a short summary and no further tool " + "calls." + ) + return ( + f"[time budget] Turn {turn}. Elapsed {_format_duration(elapsed)} of " + f"{_format_duration(budget)}; {_format_duration(remaining)} remaining. " + "This is a hard wall-clock limit: the run stops when it expires. Keep " + "any required output file up to date and reserve time to finish." + ) + # Shared by both system prompts: which tool to reach for is the same question # whoever is asking. _TOOL_GUIDANCE = ( @@ -388,6 +422,7 @@ def _run_exchange( *, max_turns: int | None = None, max_tokens: int = DEFAULT_MAX_TOKENS, + time_budget_seconds: int | None = None, reply_prefix: str = "\nAgent> ", trajectory_path: Path | None = None, ) -> RunOutcome: @@ -406,10 +441,15 @@ def _run_exchange( emitter.emit( "run.started", { - "mode": "headless" if max_turns is not None else "interactive", + "mode": ( + "headless" + if max_turns is not None or time_budget_seconds is not None + else "interactive" + ), "model": model, "max_turns": max_turns, "max_tokens": max_tokens, + "time_budget_seconds": time_budget_seconds, "producer": { "name": "nanoPyCodeAgent", "version": _package_version(), @@ -435,6 +475,7 @@ def _run_exchange( emitter=emitter, max_turns=max_turns, max_tokens=max_tokens, + time_budget_seconds=time_budget_seconds, ) except BaseException as exc: cost_reconciliation = _reconcile_costs(client, journal, emitter) @@ -488,12 +529,30 @@ def _run_model_loop( emitter: EventEmitter, max_turns: int | None, max_tokens: int, + time_budget_seconds: int | None = None, ) -> RunOutcome: """Run model replies and tool calls for an already-started Agent Run.""" turns = 0 retries = 0 retry_deadline = None + deadline = ( + time.monotonic() + time_budget_seconds + if time_budget_seconds is not None and time_budget_seconds > 0 + else None + ) while True: + if deadline is not None: + remaining = deadline - time.monotonic() + if remaining <= 0: + return "time_budget_exhausted" + turn_system = system + "\n\n" + _time_budget_note( + turn=turns + 1, + elapsed=time_budget_seconds - remaining, + budget=time_budget_seconds, + remaining=remaining, + ) + else: + turn_system = system # A spinner marks the wait for the reply; the first streamed # token replaces it with the reply prefix. A tool-only reply # streams no text, so the prefix is skipped for it entirely. @@ -515,7 +574,7 @@ def _run_model_loop( with Spinner() as spinner, client.messages.stream( model=model, max_tokens=max_tokens, - system=system, + system=turn_system, tools=TOOLS, messages=messages, ) as stream: @@ -657,6 +716,10 @@ def _run_model_loop( # Stop before running the tools: their results would only be # useful to a reply this budget can no longer pay for. return "max_turns_exhausted" + if deadline is not None and deadline - time.monotonic() <= 0: + # The reply consumed the remaining budget; its tool results would + # only be useful to a turn this run can no longer afford. + return "time_budget_exhausted" # Every tool_use block needs a matching tool_result in the next # user message, or the API rejects the request. results = [ @@ -771,6 +834,7 @@ def run_headless( *, max_turns: int = DEFAULT_MAX_TURNS, max_tokens: int | None = None, + time_budget_seconds: int | None = None, trajectory_path: Path | None = None, ) -> int: """Work ``task`` to completion without a user, and return the exit code. @@ -793,7 +857,8 @@ def run_headless( # model's prose and the echoed tool calls, nothing else. print( f"nanoPyCodeAgent v{_package_version()} — model {model}, " - f"max turns {max_turns}, max tokens {max_tokens}", + f"max turns {max_turns}, max tokens {max_tokens}, " + f"time budget {time_budget_seconds if time_budget_seconds else 'none'}", file=sys.stderr, ) @@ -806,6 +871,7 @@ def run_headless( HEADLESS_SYSTEM_PROMPT, max_turns=max_turns, max_tokens=max_tokens, + time_budget_seconds=time_budget_seconds, reply_prefix="", trajectory_path=trajectory_path, ) @@ -822,4 +888,10 @@ def run_headless( f"[stopped after {max_turns} {turns} without finishing the task]", file=sys.stderr, ) + elif outcome == "time_budget_exhausted": + print( + f"[stopped after the {time_budget_seconds}s time budget without " + "finishing the task]", + file=sys.stderr, + ) return 0 diff --git a/src/nanopycodeagent/atif.py b/src/nanopycodeagent/atif.py index af58cd1..f728dfc 100644 --- a/src/nanopycodeagent/atif.py +++ b/src/nanopycodeagent/atif.py @@ -479,6 +479,11 @@ def project_atif(entries: Sequence[JournalEntry]) -> JsonObject: if "max_tokens" in run_payload else {} ), + **( + {"time_budget_seconds": run_payload["time_budget_seconds"]} + if "time_budget_seconds" in run_payload + else {} + ), }, }, "steps": steps, diff --git a/src/nanopycodeagent/cli.py b/src/nanopycodeagent/cli.py index 93173de..be25d52 100644 --- a/src/nanopycodeagent/cli.py +++ b/src/nanopycodeagent/cli.py @@ -64,6 +64,15 @@ def _build_parser() -> argparse.ArgumentParser: f"(default: ANTHROPIC_MAX_TOKENS or {DEFAULT_MAX_TOKENS})" ), ) + parser.add_argument( + "--time-budget-seconds", + type=int, + metavar="N", + help=( + "stop a headless run after this many seconds of wall-clock time, " + "telling the model how much remains; default: no time budget" + ), + ) parser.add_argument( "--trajectory", type=Path, @@ -132,6 +141,8 @@ def main(argv: list[str] | None = None) -> int: args = parser.parse_args(argv) if args.max_turns < 1: parser.error("--max-turns must be at least 1") + if args.time_budget_seconds is not None and args.time_budget_seconds < 1: + parser.error("--time-budget-seconds must be at least 1") try: max_tokens = resolve_max_tokens(args.max_tokens) except ValueError as exc: @@ -147,5 +158,6 @@ def main(argv: list[str] | None = None) -> int: task, max_turns=args.max_turns, max_tokens=max_tokens, + time_budget_seconds=args.time_budget_seconds, trajectory_path=trajectory_path, ) diff --git a/src/nanopycodeagent/event_journal.py b/src/nanopycodeagent/event_journal.py index ae4d5c7..56c9013 100644 --- a/src/nanopycodeagent/event_journal.py +++ b/src/nanopycodeagent/event_journal.py @@ -23,7 +23,9 @@ SUPPORTED_SCHEMA_VERSIONS = frozenset({1, 2, 3}) DEFAULT_MAX_STRING_CHARS = 100_000 -type RunOutcome = Literal["completed", "max_turns_exhausted", "response_truncated"] +type RunOutcome = Literal[ + "completed", "max_turns_exhausted", "time_budget_exhausted", "response_truncated" +] EVENT_TYPES = frozenset( { @@ -440,7 +442,7 @@ def _validate_native_payload(event_type: str, payload: JsonObject) -> None: _validate_tool_error(payload.get("error")) elif event_type == "run.completed": if payload["outcome"] not in { - "completed", "max_turns_exhausted", "response_truncated" + "completed", "max_turns_exhausted", "time_budget_exhausted", "response_truncated" }: raise ValueError("run.completed.outcome is unsupported") _validate_cost_reconciliation(payload, event_type) diff --git a/tests/test_time_budget.py b/tests/test_time_budget.py new file mode 100644 index 0000000..74880f0 --- /dev/null +++ b/tests/test_time_budget.py @@ -0,0 +1,151 @@ +"""A headless wall-clock budget must reach the model and stop the run. + +Unlike ``max_turns`` (a count of replies), the time budget is what the harness +actually enforces. The loop has to tell the model how much time is left and +stop on its own before that harness timeout kills the process with nothing +written. +""" + +import pytest + +from nanopycodeagent import agent, cli, settings +from nanopycodeagent.event_journal import EventJournal + +from helpers import ( + FakeClient, + FakeMessages, + FakeStream, + patch_client, + text_block, + tool_use_block, +) + + +class FakeTime: + def __init__(self, start=1000.0): + self.now = start + + def monotonic(self): + return self.now + + def perf_counter_ns(self): + return int(self.now * 1_000_000_000) + + def advance(self, seconds): + self.now += seconds + + +class AdvancingFakeMessages(FakeMessages): + """Advance a fake clock every time a model reply is requested.""" + + def __init__(self, script, clock, seconds_per_call): + super().__init__(script) + self.clock = clock + self.seconds_per_call = seconds_per_call + + def stream(self, **kwargs): + self.clock.advance(self.seconds_per_call) + return super().stream(**kwargs) + + +def _journal_entries(): + paths = list((settings.SETTINGS_PATH.parent / "journals").glob("*.jsonl")) + assert len(paths) == 1 + return EventJournal.replay(paths[0]) + + +def _fake_bash(executions): + def run_bash(command): + executions.append(command) + return "ok", False + + return run_bash + + +def test_time_budget_is_injected_and_stops_the_run(monkeypatch, capsys): + clock = FakeTime() + monkeypatch.setattr(agent, "time", clock) + executions = [] + monkeypatch.setattr(agent, "run_bash", _fake_bash(executions)) + + replies = [ + FakeStream([tool_use_block("call-1", "echo one")], stop_reason="tool_use"), + FakeStream([tool_use_block("call-2", "echo two")], stop_reason="tool_use"), + FakeStream([tool_use_block("call-3", "echo three")], stop_reason="tool_use"), + FakeStream([text_block("done")], stop_reason="end_turn"), + ] + # 450s consumed per reply against a 1000s budget: the first two replies + # still have room, the third lands in the reserved finalization window, + # and by the fourth the deadline has already passed. + messages = AdvancingFakeMessages(replies, clock, 450.0) + patch_client(monkeypatch, FakeClient(messages)) + + assert ( + cli.main( + [ + "-p", + "finish the task", + "--max-turns", + "10", + "--time-budget-seconds", + "1000", + ] + ) + == 0 + ) + + # The third reply crossed the deadline, so its tools never run and a + # fourth reply is never requested. + assert len(messages.calls) == 3 + assert executions == ["echo one", "echo two"] + + assert "Elapsed 0:00 of 16:40; 16:40 remaining" in messages.kwargs[0]["system"] + assert "Elapsed 7:30 of 16:40; 9:10 remaining" in messages.kwargs[1]["system"] + final_system = messages.kwargs[2]["system"] + assert "Only 1:40 of 16:40 left" in final_system + assert "Stop investigating now" in final_system + + captured = capsys.readouterr() + assert "1000s time budget" in captured.err + assert "stopped after 10 turns" not in captured.err + + entries = _journal_entries() + assert entries[0].payload["time_budget_seconds"] == 1000 + assert entries[-1].type == "run.completed" + assert entries[-1].payload["outcome"] == "time_budget_exhausted" + + +def test_without_a_budget_the_system_prompt_is_unchanged(monkeypatch): + clock = FakeTime() + monkeypatch.setattr(agent, "time", clock) + monkeypatch.setattr(agent, "run_bash", _fake_bash([])) + messages = FakeMessages([FakeStream([text_block("done")], stop_reason="end_turn")]) + patch_client(monkeypatch, FakeClient(messages)) + + assert cli.main(["-p", "just answer", "--max-turns", "5"]) == 0 + + assert "[time budget]" not in messages.kwargs[0]["system"] + entries = _journal_entries() + assert entries[0].payload["time_budget_seconds"] is None + assert entries[-1].payload["outcome"] == "completed" + + +@pytest.mark.parametrize("value", ["0", "-3"]) +def test_non_positive_budget_is_a_usage_error(value, monkeypatch, capsys): + messages = FakeMessages([]) + patch_client(monkeypatch, FakeClient(messages)) + with pytest.raises(SystemExit) as excinfo: + cli.main(["-p", "task", "--time-budget-seconds", value]) + assert excinfo.value.code == cli.EXIT_USAGE + assert "at least 1" in capsys.readouterr().err + + +def test_finalization_note_escalates_below_the_reserve(): + early = agent._time_budget_note( + turn=1, elapsed=10, budget=1000, remaining=990 + ) + late = agent._time_budget_note( + turn=9, elapsed=990, budget=1000, remaining=10 + ) + assert "remaining" in early and "Stop investigating now" not in early + assert "Stop investigating now" in late From 5470c21c91388f373111ee1ca7e1242f572801c0 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Sun, 27 Sep 2026 00:41:40 +0800 Subject: [PATCH 03/12] docs: record time-budget experiment Experiment 2 adds a wall-clock budget, per-turn budget injection, and stop-on-deadline on top of Experiment 1. Both tasks passed and self-terminated cleanly with native trajectories. Also records that mutating the system prompt each turn broke prompt caching and raised cost about 9x. --- .../tb21-exp2-timebudget-20260926.json | 92 +++++++++++++++++++ docs/dev_notes/en/0.8.x.md | 85 +++++++++++++++++ docs/dev_notes/zh-CN/0.8.x.md | 72 +++++++++++++++ 3 files changed, 249 insertions(+) create mode 100644 benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json diff --git a/benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json b/benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json new file mode 100644 index 0000000..e22b60d --- /dev/null +++ b/benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json @@ -0,0 +1,92 @@ +{ + "schema_version": 1, + "generated_at": "2026-09-26T16:41:12.564300+00:00", + "purpose": "Experiment 2: add an agent wall-clock time budget (budget injection plus stop-on-deadline) on top of Experiment 1, for the two tasks that still scored reward 0 in tb21-v41flash-6task-20260925.", + "agent_ref": "eb77fd578a3b08582f3f6958858c0756b66d58dc", + "agent_version": "0.8.1.dev34+geb77fd578", + "model": "openrouter/deepseek/deepseek-v4.1-flash", + "dataset": "terminal-bench/terminal-bench-2-1", + "dataset_ref": "sha256:7d7bdc1cbedad549fc1140404bd4dc45e5fd0ea7c4186773687d177ad3a0699a", + "selection": "The two tasks that still scored reward 0 in tb21-v41flash-6task-20260925 (caffe-cifar-10, mteb-leaderboard) with their pinned task refs.", + "configuration": { + "max_turns": 100, + "time_budget_seconds": 3420, + "previous_max_turns": 50, + "max_tokens": 65536, + "project_default_max_tokens": 32768, + "agent_execution_limit_seconds": 3480, + "finalization_grace_seconds": 120, + "harbor_outer_limit_seconds": null, + "agent_setup_limit_seconds": null, + "timeout_compliance": "Leaderboard-compliant: no Harbor agent/verifier timeout override and no setup timeout override; the effective agent timeout is the task-native 3600s. The agent self-limits at 3480s inside it.", + "maximum_simultaneous_trials": 2, + "automatic_harbor_retries": 0, + "attempts_per_task": 1, + "verifier_limits": "Native task deadlines", + "dependency_constraints": [ + "anthropic==1.5.0", + "httpx==0.28.1", + "httpx2==2.12.0", + "pydantic==2.13.5" + ] + }, + "baseline": { + "job": "tb21-v41flash-6task-20260925", + "max_turns": 50, + "note": "Both re-run tasks scored reward 0 there by exhausting 50 turns (max_turns_exhausted)." + }, + "totals": { + "planned": 2, + "finished": 2, + "passed": 2, + "scored_zero": 0, + "unscored": 0, + "model_attempts": 111, + "input_tokens": 2919588, + "cache_tokens": 1408, + "output_tokens": 57873, + "cost_usd": 0.856032128 + }, + "trials": [ + { + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "reward": 1.0, + "outcome": "completed", + "duration_seconds": 2656.9, + "model_replies": 61, + "tool_calls": 70, + "input_tokens": 1326596, + "cache_tokens": 1408, + "output_tokens": 27327, + "cost_usd": 0.374407898, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "started_at": "2026-09-26T15:43:24.927Z" + }, + { + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "reward": 1.0, + "outcome": "completed", + "duration_seconds": 864.7, + "model_replies": 50, + "tool_calls": 59, + "input_tokens": 1592992, + "cache_tokens": 0, + "output_tokens": 30546, + "cost_usd": 0.48162423, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "started_at": "2026-09-26T15:42:58.375Z" + } + ], + "comparison_limits": "Single attempt per task; routing, sampling, cache and provider backend are uncontrolled, so this is a targeted diagnostic, not a controlled estimate of pass-rate improvement.", + "artifacts": { + "raw_trials": "jobs/tb21-exp2-timebudget-20260926", + "record": "jobs/tb21-exp2-timebudget-20260926-record", + "note": "Raw trials and record live in Git-ignored jobs/ and are not uploaded." + } +} diff --git a/docs/dev_notes/en/0.8.x.md b/docs/dev_notes/en/0.8.x.md index 9fb01b2..4cd4bed 100644 --- a/docs/dev_notes/en/0.8.x.md +++ b/docs/dev_notes/en/0.8.x.md @@ -1197,3 +1197,88 @@ Git-ignored `jobs/` and were not uploaded. Version control stores the structured result [`tb21-exp1-turn100-20260926.json`](../../../benchmarks/harbor/results/tb21-exp1-turn100-20260926.json) and this summary. + +### Experiment 2: time budget, budget injection, and stop-on-deadline + +**Run on 2026-09-26.** Experiment 1 showed that raising max_turns lets both tasks +pass, but `caffe-cifar-10` was still killed externally (`forced_kill=true`, the +trajectory reconstructed). This run adds three things on top of Experiment 1 and +changes nothing else: + +- **Time budget**: a new `--time-budget-seconds` (Harbor kwarg + `time_budget_seconds`). This run uses 3420s, inside the task-native 3600s and + before the supervisor's 3480s self-limit. +- **Budget injection**: the system prompt carries the remaining time each turn, + and inside the finalization window (`max(15% of the budget, 180s)`) it + escalates to "stop investigating, write the output, then finish". +- **Stop-on-deadline**: when the budget runs out the run returns a new + `time_budget_exhausted` outcome, stopping before the next model call or tool + execution, and records it in the Journal and ATIF. + +| Item | This run | +| --- | --- | +| Job | `tb21-exp2-timebudget-20260926` | +| Agent source | branch `experiment/max-turns-and-time-budget`, commit `eb77fd578`; wheel `0.8.1.dev34+geb77fd578` | +| Model | `openrouter/deepseek/deepseek-v4.1-flash` | +| Time limits | task-native agent 3600s; agent budget 3420s; supervisor self-limits to 3480s + 120s finalization; **no Harbor timeout overridden** | +| Everything else | at most 100 replies per task, at most 65536 tokens per reply, concurrency 2, one attempt, zero retries — same as Experiment 1 | + +**Result: both tasks passed, and this time both wound down on their own.** + +| Task | Reward | Terminal state | Model replies | Duration | Cost (USD) | +| --- | ---: | --- | ---: | ---: | ---: | +| `mteb-leaderboard` | 1 | `completed` | 50 | 865 s | 0.481624230 | +| `caffe-cifar-10` | 1 | `completed` | 61 | 2657 s | 0.374407898 | + +- `mteb-leaderboard`: finished in 50 replies and 865s, **twice as fast** as + Experiment 1's 70 replies and 1666s; its final message names the answer + `GritLM/GritLM-7B` and states it wrote `/app/result.txt`. +- `caffe-cifar-10`: 61 replies and 2657s, **stopping on its own** — the supervisor + reports `deadline_reached=false`, `forced_kill=false`, + `trajectory_source=native`, in contrast to Experiment 1, where it kept building + pycaffe after training, was force-killed, and its trajectory had to be + reconstructed. Its final message is "All requirements are met", with exactly 500 + training iterations and 55.17% test accuracy. + +**The two experiments side by side (two tasks each, one trial each):** + +| Metric | Experiment 1 (max_turns=100) | Experiment 2 (+ 3420s budget) | +| --- | ---: | ---: | +| Passed | 2 / 2 | 2 / 2 | +| Clean `completed` | 1 / 2 | **2 / 2** | +| Deadline hit / forced kill | 1 (caffe) | **0** | +| Trajectory reconstructed | 1 (caffe) | **0** | +| Total model calls | 139 | 111 | +| Input tokens | 340,804 | 2,919,588 | +| Cache tokens | 4,289,280 | 1,408 | +| Total cost (USD) | 0.094 | **0.856** | + +**Finding 1: the time budget makes the agent finish on its own.** Both tasks +stopped before the budget ran out, with `deadline_reached` and `forced_kill` both +false and native trajectories. This is exactly the "does the work but will not +stop" gap that Experiment 1 exposed. + +**Finding 2: rewriting the system prompt every turn broke prompt caching and +raised cost about 9×.** Experiment 2 used about 8.6× the input tokens of +Experiment 1, cache tokens fell from 4.29M to 1.4K, and total cost rose from +$0.094 to $0.856. Appending the budget note to the system prompt makes **the +prefix different every turn**, so the provider's longest-common-prefix cache never +hits and every turn re-reads the whole context. This is a real design flaw: the +fix is to **keep the system prompt fixed and append the budget note as a user/tool +message at the tail** (a tail change does not invalidate the prefix), or to inject +only when crossing a threshold. This run records the flaw as-is; the fix is left +for the next round. + +**Evidence boundary.** Still one trial per task, with routing, sampling, cache, +and backend uncontrolled; "faster and cleaner" is a targeted signal, not a +controlled estimate, and run-to-run variance cannot be ruled out. The only thing +directly attributable is the mechanism itself: `deadline_reached=false`, +`forced_kill=false`, and `trajectory_source=native` are direct evidence that the +agent stopped itself. + +**Artifacts.** Raw trials are under `jobs/tb21-exp2-timebudget-20260926/`, with the +experiment record under `jobs/tb21-exp2-timebudget-20260926-record/`; these live +in Git-ignored `jobs/` and were not uploaded. Version control stores the +structured result +[`tb21-exp2-timebudget-20260926.json`](../../../benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json) +and this summary. diff --git a/docs/dev_notes/zh-CN/0.8.x.md b/docs/dev_notes/zh-CN/0.8.x.md index f12927a..eab3240 100644 --- a/docs/dev_notes/zh-CN/0.8.x.md +++ b/docs/dev_notes/zh-CN/0.8.x.md @@ -1599,3 +1599,75 @@ Terminal-Bench / Harbor 对 agent 的硬约束是**墙钟时间**(本题原生 版本管理保存结构化结果 [`tb21-exp1-turn100-20260926.json`](../../../benchmarks/harbor/results/tb21-exp1-turn100-20260926.json) 与本节汇总。 + +### 实验二:时间预算、预算注入与到点停止 + +**2026-09-26 实跑。** 实验一证明"放宽轮数"能让两题通过,但 `caffe-cifar-10` +仍是被外部强杀(`forced_kill=true`、轨迹靠重建)。本轮在实验一的基础上加三样东西, +其余不变: + +- **时间预算**:新增 `--time-budget-seconds`(Harbor 侧 kwarg + `time_budget_seconds`)。本轮取 3420 秒,落在题目原生 3600 秒之内、supervisor + 3480 秒自限之前。 +- **预算注入**:每轮把剩余时间写进 system prompt;进入收尾窗口 + (`max(15% 预算, 180 秒)`)后升级为"立即停止调研、写出产物、然后收尾"的强提示。 +- **到点停止**:预算耗尽时返回新终态 `time_budget_exhausted`,在下一次调用或执行 + 工具之前停下,并写进 Journal 与 ATIF。 + +| 项目 | 本轮配置 | +| --- | --- | +| Job | `tb21-exp2-timebudget-20260926` | +| Agent 源码 | 分支 `experiment/max-turns-and-time-budget`,提交 `eb77fd578`;wheel `0.8.1.dev34+geb77fd578` | +| Model | `openrouter/deepseek/deepseek-v4.1-flash` | +| 时限 | 题目原生 agent 3600 秒;agent 预算 3420 秒;supervisor 自限 3480 秒 + 120 秒收尾;**未覆盖任何 Harbor timeout** | +| 其余 | 每题最多 100 次回复、每次最多 65536 tokens、并发 2、每题 1 次、重试 0,与实验一一致 | + +**结果:2 题全部通过,而且这次两题都自己干净收尾。** + +| 题目 | Reward | 终态 | 模型回复 | 时长 | 费用(USD) | +| --- | ---: | --- | ---: | ---: | ---: | +| `mteb-leaderboard` | 1 | `completed` | 50 | 865 s | 0.481624230 | +| `caffe-cifar-10` | 1 | `completed` | 61 | 2657 s | 0.374407898 | + +- `mteb-leaderboard`:50 轮、865 秒完成,比实验一的 70 轮、1666 秒**快了一倍**, + 最终消息明确写出答案 `GritLM/GritLM-7B` 并说明已写入 `/app/result.txt`。 +- `caffe-cifar-10`:61 轮、2657 秒**主动收尾**——supervisor 显示 + `deadline_reached=false`、`forced_kill=false`、`trajectory_source=native`, + 与实验一"训练完成后继续构建 pycaffe、最终被强杀、轨迹靠重建"形成对照。 + 最终消息是"All requirements are met",训练恰好 500 步、测试精度 55.17%。 + +**两轮实验并列对比(同为 2 题、各 1 次):** + +| 指标 | 实验一(max_turns=100) | 实验二(+ 3420s 预算) | +| --- | ---: | ---: | +| 通过 | 2 / 2 | 2 / 2 | +| 干净 `completed` | 1 / 2 | **2 / 2** | +| 触发截止/强杀 | 1(caffe) | **0** | +| 轨迹需重建 | 1(caffe) | **0** | +| 总模型调用 | 139 | 111 | +| input tokens | 340,804 | 2,919,588 | +| cache tokens | 4,289,280 | 1,408 | +| 总费用(USD) | 0.094 | **0.856** | + +**发现一:时间预算让 agent 自己收尾。** 两题都在预算耗尽之前主动结束, +`deadline_reached` 与 `forced_kill` 全为 false,轨迹全部是原生的。这正好补上了 +实验一暴露的"能做完但不会停"。 + +**发现二:逐轮改写 system prompt 破坏了 prompt 缓存,费用约涨 9 倍。** +实验二的 input tokens 是实验一的约 8.6 倍,cache tokens 从 4.29M 掉到 1.4K, +总费用从 $0.094 涨到 $0.856。原因是把预算提示拼进 system prompt 后,**每轮前缀 +都不同**,provider 的最长公共前缀缓存完全失效,每轮都要重新读入全部上下文。 +这是一个真实的设计缺陷,应改为**保持 system prompt 固定,把预算提示作为对话中 +的一条 user/tool 消息追加在尾部**(尾部变化不影响前缀缓存),或只在跨阈值时注入。 +本次先如实记录,修复方案留待下一轮。 + +**证据边界。** 每题仍只跑 1 次,路由、采样、缓存与后端都未固定;"更快、更干净" +是针对性信号,不是受控估计,也不能排除单次运行波动。唯一能直接归因的是机制本身: +`deadline_reached=false`、`forced_kill=false`、`trajectory_source=native` 是 +"agent 自己停"的直接证据。 + +**产物位置。** 原始 trial 在 `jobs/tb21-exp2-timebudget-20260926/`,实验记录在 +`jobs/tb21-exp2-timebudget-20260926-record/`;位于 Git 忽略的 `jobs/`,未上传。 +版本管理保存结构化结果 +[`tb21-exp2-timebudget-20260926.json`](../../../benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json) +与本节汇总。 From 92ddfc35dd767dabcbd39f04040874d159c47277 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Sun, 27 Sep 2026 00:42:20 +0800 Subject: [PATCH 04/12] docs: correct the time-budget cost finding Attribute the cache collapse to the per-turn system-prompt change using the DeepInfra before/after comparison, and record the concurrent provider-routing shift as a confounder in the structured results. --- .../results/tb21-exp1-turn100-20260926.json | 29 +++++++++++++++ .../tb21-exp2-timebudget-20260926.json | 35 +++++++++++++++++++ docs/dev_notes/en/0.8.x.md | 32 +++++++++++------ docs/dev_notes/zh-CN/0.8.x.md | 24 +++++++++---- 4 files changed, 103 insertions(+), 17 deletions(-) diff --git a/benchmarks/harbor/results/tb21-exp1-turn100-20260926.json b/benchmarks/harbor/results/tb21-exp1-turn100-20260926.json index 19a770b..c357ace 100644 --- a/benchmarks/harbor/results/tb21-exp1-turn100-20260926.json +++ b/benchmarks/harbor/results/tb21-exp1-turn100-20260926.json @@ -87,5 +87,34 @@ "raw_trials": "jobs/tb21-exp1-turn100-20260926", "record": "jobs/tb21-exp1-turn100-20260926-record", "note": "Raw trials and record live in Git-ignored jobs/ and are not uploaded." + }, + "provider_routing": { + "by_provider": { + "DeepInfra": { + "replies": 121, + "input_tokens": 140692, + "cache_read_tokens": 3523072, + "output_tokens": 66677 + }, + "Parasail": { + "replies": 2, + "input_tokens": 100123, + "cache_read_tokens": 0, + "output_tokens": 582 + }, + "(unresolved)": { + "replies": 15, + "input_tokens": 98130, + "cache_read_tokens": 766208, + "output_tokens": 6293 + }, + "Together": { + "replies": 1, + "input_tokens": 1859, + "cache_read_tokens": 0, + "output_tokens": 247 + } + }, + "note": "Routing is uncontrolled. DeepInfra was the only provider reporting cache reads in both runs; its cache reads per reply collapsed from ~29116 (Experiment 1) to ~61 (Experiment 2), consistent with the per-turn system-prompt change invalidating the prefix cache." } } diff --git a/benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json b/benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json index e22b60d..f3317ad 100644 --- a/benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json +++ b/benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json @@ -88,5 +88,40 @@ "raw_trials": "jobs/tb21-exp2-timebudget-20260926", "record": "jobs/tb21-exp2-timebudget-20260926-record", "note": "Raw trials and record live in Git-ignored jobs/ and are not uploaded." + }, + "provider_routing": { + "by_provider": { + "DeepInfra": { + "replies": 23, + "input_tokens": 209149, + "cache_read_tokens": 1408, + "output_tokens": 9613 + }, + "Together": { + "replies": 48, + "input_tokens": 1477246, + "cache_read_tokens": 0, + "output_tokens": 22297 + }, + "Relace": { + "replies": 10, + "input_tokens": 209067, + "cache_read_tokens": 0, + "output_tokens": 5079 + }, + "Parasail": { + "replies": 29, + "input_tokens": 1021931, + "cache_read_tokens": 0, + "output_tokens": 20280 + }, + "StreamLake": { + "replies": 1, + "input_tokens": 2195, + "cache_read_tokens": 0, + "output_tokens": 604 + } + }, + "note": "Routing is uncontrolled. DeepInfra was the only provider reporting cache reads in both runs; its cache reads per reply collapsed from ~29116 (Experiment 1) to ~61 (Experiment 2), consistent with the per-turn system-prompt change invalidating the prefix cache." } } diff --git a/docs/dev_notes/en/0.8.x.md b/docs/dev_notes/en/0.8.x.md index 4cd4bed..c8e02cf 100644 --- a/docs/dev_notes/en/0.8.x.md +++ b/docs/dev_notes/en/0.8.x.md @@ -1258,16 +1258,28 @@ stopped before the budget ran out, with `deadline_reached` and `forced_kill` bot false and native trajectories. This is exactly the "does the work but will not stop" gap that Experiment 1 exposed. -**Finding 2: rewriting the system prompt every turn broke prompt caching and -raised cost about 9×.** Experiment 2 used about 8.6× the input tokens of -Experiment 1, cache tokens fell from 4.29M to 1.4K, and total cost rose from -$0.094 to $0.856. Appending the budget note to the system prompt makes **the -prefix different every turn**, so the provider's longest-common-prefix cache never -hits and every turn re-reads the whole context. This is a real design flaw: the -fix is to **keep the system prompt fixed and append the budget note as a user/tool -message at the tail** (a tail change does not invalidate the prefix), or to inject -only when crossing a threshold. This run records the flaw as-is; the fix is left -for the next round. +**Finding 2: rewriting the system prompt every turn breaks the prefix cache; the +total cost rose, but routing confounds it.** On the surface, Experiment 2 used +about 8.6× the input tokens of Experiment 1 (2.92M vs 0.34M), cache tokens fell +from 4.29M to 1.4K, and total cost rose from $0.094 to $0.856. Split by provider, +there are two causes: + +1. **The system prompt changes every turn, so the prefix cache misses.** DeepInfra + is the only provider that reports cache reads consistently in both runs. In + Experiment 1 its 121 calls averaged 29,116 cached tokens each; in Experiment 2 + its 23 calls averaged 61. On the same provider the prefix cache essentially + vanished, consistent with a system prompt that invalidates the + longest-common-prefix cache every turn. +2. **Routing shifted (uncontrolled).** Experiment 1 sent 121 of 139 calls to + DeepInfra; Experiment 2 spread its 111 calls across Together (48), Parasail + (29), DeepInfra (23), and Relace (10). Together, Parasail, and Relace reported + no cache reads in either run. So the 8.6× cost increase **cannot be attributed + to the injection alone**; the routing change is a confounder on top of it. + +Either way the fix is the same: **keep the system prompt fixed and append the +budget note as a user/tool message at the tail** (a tail change does not +invalidate the prefix), or inject only when crossing a threshold. This run +records the flaw as-is; the fix is left for the next round. **Evidence boundary.** Still one trial per task, with routing, sampling, cache, and backend uncontrolled; "faster and cleaner" is a targeted signal, not a diff --git a/docs/dev_notes/zh-CN/0.8.x.md b/docs/dev_notes/zh-CN/0.8.x.md index eab3240..b70bb5e 100644 --- a/docs/dev_notes/zh-CN/0.8.x.md +++ b/docs/dev_notes/zh-CN/0.8.x.md @@ -1653,13 +1653,23 @@ Terminal-Bench / Harbor 对 agent 的硬约束是**墙钟时间**(本题原生 `deadline_reached` 与 `forced_kill` 全为 false,轨迹全部是原生的。这正好补上了 实验一暴露的"能做完但不会停"。 -**发现二:逐轮改写 system prompt 破坏了 prompt 缓存,费用约涨 9 倍。** -实验二的 input tokens 是实验一的约 8.6 倍,cache tokens 从 4.29M 掉到 1.4K, -总费用从 $0.094 涨到 $0.856。原因是把预算提示拼进 system prompt 后,**每轮前缀 -都不同**,provider 的最长公共前缀缓存完全失效,每轮都要重新读入全部上下文。 -这是一个真实的设计缺陷,应改为**保持 system prompt 固定,把预算提示作为对话中 -的一条 user/tool 消息追加在尾部**(尾部变化不影响前缀缓存),或只在跨阈值时注入。 -本次先如实记录,修复方案留待下一轮。 +**发现二:逐轮改写 system prompt 破坏前缀缓存;总费用上升但被路由混淆。** +表面上,实验二的 input tokens 是实验一的约 8.6 倍(2.92M 对 0.34M),cache +tokens 从 4.29M 掉到 1.4K,总费用从 $0.094 涨到 $0.856。按 provider 拆开后有两个 +原因: + +1. **system prompt 每轮都变,前缀缓存失效。** DeepInfra 是两次运行里唯一稳定上报 + 缓存的 provider。实验一 121 次 DeepInfra 调用平均每次命中 29,116 个缓存 tokens; + 实验二 23 次 DeepInfra 调用平均只命中 61 个。同一 provider 上前缀缓存几乎归零, + 与"每轮改写 system prompt 使最长公共前缀失效"一致。 +2. **路由发生偏移(未受控)。** 实验一 139 次调用里 121 次落在 DeepInfra;实验二 + 111 次里 Together 48、Parasail 29、DeepInfra 23、Relace 10。Together/Parasail/ + Relace 在两次运行里都不上报缓存读取。因此总费用的 8.6 倍**不能单独归因于注入**, + 路由变化是叠加的混淆因素。 + +无论哪种原因,修复方向一致:**保持 system prompt 固定,把预算提示作为对话尾部 +的一条 user/tool 消息追加**(尾部变化不影响前缀缓存),或只在跨阈值时注入。本次先 +如实记录,修复方案留待下一轮。 **证据边界。** 每题仍只跑 1 次,路由、采样、缓存与后端都未固定;"更快、更干净" 是针对性信号,不是受控估计,也不能排除单次运行波动。唯一能直接归因的是机制本身: From 83e5c064d4c6e421fc08a90bbd84d2d23b6bb3cf Mon Sep 17 00:00:00 2001 From: minixalpha Date: Sun, 27 Sep 2026 21:45:43 +0800 Subject: [PATCH 05/12] fix: append the time-budget reminder to the conversation tail Rewrite the per-turn reminder into the most recent user message instead of the system prompt. A changing system prompt made the first tokens of every request differ, defeating the provider's prefix cache; appending to the tail keeps each request an extension of the previous one so the cached prefix survives. --- docs/changelogs/0.8.x.md | 4 +++- src/nanopycodeagent/agent.py | 34 ++++++++++++++++++++++++++-------- tests/test_time_budget.py | 30 +++++++++++++++++++++++++----- 3 files changed, 54 insertions(+), 14 deletions(-) diff --git a/docs/changelogs/0.8.x.md b/docs/changelogs/0.8.x.md index 6007a9a..e4e3325 100644 --- a/docs/changelogs/0.8.x.md +++ b/docs/changelogs/0.8.x.md @@ -14,7 +14,9 @@ All notable changes in the **0.8.x** release series are documented here. Tell the model how much time remains each turn, escalate to a stop-and-write warning inside a reserved finalization window, and stop with a `time_budget_exhausted` outcome before the harness timeout can kill the - process. Record the budget in startup output, Event Journals, and ATIF. + process. Append the reminder to the tail of the conversation so the system + prompt stays fixed and provider prefix caching keeps working. Record the + budget in startup output, Event Journals, and ATIF. ### Fixed - Return correctable tool errors for missing or incorrectly typed arguments and diff --git a/src/nanopycodeagent/agent.py b/src/nanopycodeagent/agent.py index 9f4f863..82551e1 100644 --- a/src/nanopycodeagent/agent.py +++ b/src/nanopycodeagent/agent.py @@ -118,6 +118,23 @@ def _time_budget_note( "any required output file up to date and reserve time to finish." ) + +def _append_budget_note(messages: list[MessageParam], note: str) -> None: + """Append a wall-clock reminder to the tail of the conversation. + + The reminder goes into the most recent user message — the initial task, or + the tool results — rather than the system prompt. Rewriting the system + prompt each turn changes the very first tokens of every request and defeats + the provider's prefix cache; appending to the tail keeps each request an + extension of the previous one, so the cached prefix survives. + """ + last = messages[-1] + content = last["content"] + if isinstance(content, str): + last["content"] = f"{content}\n\n{note}" + else: + content.append({"type": "text", "text": note}) + # Shared by both system prompts: which tool to reach for is the same question # whoever is asking. _TOOL_GUIDANCE = ( @@ -545,14 +562,15 @@ def _run_model_loop( remaining = deadline - time.monotonic() if remaining <= 0: return "time_budget_exhausted" - turn_system = system + "\n\n" + _time_budget_note( - turn=turns + 1, - elapsed=time_budget_seconds - remaining, - budget=time_budget_seconds, - remaining=remaining, + _append_budget_note( + messages, + _time_budget_note( + turn=turns + 1, + elapsed=time_budget_seconds - remaining, + budget=time_budget_seconds, + remaining=remaining, + ), ) - else: - turn_system = system # A spinner marks the wait for the reply; the first streamed # token replaces it with the reply prefix. A tool-only reply # streams no text, so the prefix is skipped for it entirely. @@ -574,7 +592,7 @@ def _run_model_loop( with Spinner() as spinner, client.messages.stream( model=model, max_tokens=max_tokens, - system=turn_system, + system=system, tools=TOOLS, messages=messages, ) as stream: diff --git a/tests/test_time_budget.py b/tests/test_time_budget.py index 74880f0..67660bb 100644 --- a/tests/test_time_budget.py +++ b/tests/test_time_budget.py @@ -54,6 +54,20 @@ def _journal_entries(): return EventJournal.replay(paths[0]) +def _texts(messages): + """All text carried by a snapshot of the message list.""" + out = [] + for message in messages: + content = message["content"] + if isinstance(content, str): + out.append(content) + else: + for block in content: + if isinstance(block, dict) and block.get("type") == "text": + out.append(block["text"]) + return "\n".join(out) + + def _fake_bash(executions): def run_bash(command): executions.append(command) @@ -99,11 +113,16 @@ def test_time_budget_is_injected_and_stops_the_run(monkeypatch, capsys): assert len(messages.calls) == 3 assert executions == ["echo one", "echo two"] - assert "Elapsed 0:00 of 16:40; 16:40 remaining" in messages.kwargs[0]["system"] - assert "Elapsed 7:30 of 16:40; 9:10 remaining" in messages.kwargs[1]["system"] - final_system = messages.kwargs[2]["system"] - assert "Only 1:40 of 16:40 left" in final_system - assert "Stop investigating now" in final_system + # The reminder must not touch the system prompt: keeping it constant is what + # lets the provider reuse its prefix cache. It rides at the tail instead. + systems = [call["system"] for call in messages.kwargs] + assert len(set(systems)) == 1 + assert "[time budget]" not in systems[0] + assert "Elapsed 0:00 of 16:40; 16:40 remaining" in _texts(messages.calls[0]) + assert "Elapsed 7:30 of 16:40; 9:10 remaining" in _texts(messages.calls[1]) + final_text = _texts(messages.calls[2]) + assert "Only 1:40 of 16:40 left" in final_text + assert "Stop investigating now" in final_text captured = capsys.readouterr() assert "1000s time budget" in captured.err @@ -125,6 +144,7 @@ def test_without_a_budget_the_system_prompt_is_unchanged(monkeypatch): assert cli.main(["-p", "just answer", "--max-turns", "5"]) == 0 assert "[time budget]" not in messages.kwargs[0]["system"] + assert "[time budget]" not in _texts(messages.calls[0]) entries = _journal_entries() assert entries[0].payload["time_budget_seconds"] is None assert entries[-1].payload["outcome"] == "completed" From 2ea491ffefff49fdc8ec951d7e3b20487b1a71dd Mon Sep 17 00:00:00 2001 From: minixalpha Date: Sun, 27 Sep 2026 23:01:31 +0800 Subject: [PATCH 06/12] fix: address time-budget review findings - Recheck the wall-clock deadline before every tool in a reply, not only once before the batch, so a later tool cannot start after the budget expires. - Record each injected budget reminder as an input.injected event and project it into ATIF, so the behavior-changing input reaches the trajectory. - Bump the Event Journal schema to v4 for time_budget_exhausted and input.injected, following the response_truncated precedent, and update the protocol docs. --- .../harbor/tests/test_atif_compatibility.py | 30 ++++++ docs/changelogs/0.8.x.md | 13 ++- docs/dev_docs/README.md | 5 +- docs/dev_docs/en/event-journal-protocol-v3.md | 10 +- docs/dev_docs/en/event-journal-protocol-v4.md | 91 ++++++++++++++++ .../zh-CN/event-journal-protocol-v3.md | 6 +- .../zh-CN/event-journal-protocol-v4.md | 73 +++++++++++++ src/nanopycodeagent/agent.py | 47 +++++--- src/nanopycodeagent/atif.py | 15 ++- src/nanopycodeagent/event_journal.py | 21 +++- tests/test_event_journal.py | 26 ++++- tests/test_stream_recovery.py | 2 +- tests/test_time_budget.py | 102 ++++++++++++++++++ tests/test_truncation.py | 13 ++- 14 files changed, 413 insertions(+), 41 deletions(-) create mode 100644 docs/dev_docs/en/event-journal-protocol-v4.md create mode 100644 docs/dev_docs/zh-CN/event-journal-protocol-v4.md diff --git a/benchmarks/harbor/tests/test_atif_compatibility.py b/benchmarks/harbor/tests/test_atif_compatibility.py index af10875..4456078 100644 --- a/benchmarks/harbor/tests/test_atif_compatibility.py +++ b/benchmarks/harbor/tests/test_atif_compatibility.py @@ -16,6 +16,36 @@ def test_projector_output_passes_harbor_atif_validator(): assert validator.validate(trajectory), validator.get_errors() +def test_time_budget_v4_passes_harbor_atif_validator(tmp_path): + fixture = Path(__file__).parent / "fixtures" / "atif-journal-v1.jsonl" + note = "[time budget] Stop investigating now. Write your best answer to the output file." + with EventJournal.create("run-budgeted", directory=tmp_path) as journal: + for entry in EventJournal.replay(fixture): + if entry.type == "model.started": + journal.append(NativeEvent("input.injected", { + "model_call_id": entry.payload["model_call_id"], + "content": note, + "reason": "time_budget", + "source_timestamp": entry.payload["source_timestamp"], + })) + payload = entry.payload + if entry.type == "run.completed": + payload = payload | {"outcome": "time_budget_exhausted"} + journal.append(NativeEvent(entry.type, payload)) + + entries = EventJournal.replay(journal.path) + assert all(entry.schema_version == 4 for entry in entries) + trajectory = project_atif(entries) + validator = TrajectoryValidator() + assert validator.validate(trajectory), validator.get_errors() + assert trajectory["extra"]["terminal"]["outcome"] == "time_budget_exhausted" + reminder = trajectory["steps"][1] + assert reminder["source"] == "user" + assert reminder["message"] == note + assert reminder["extra"]["injected"] is True + assert reminder["extra"]["model_call_id"] == trajectory["steps"][2]["extra"]["model_call_id"] + + @pytest.mark.parametrize("resolved", [False, True]) def test_recovered_stream_v3_passes_harbor_atif_validator(tmp_path, resolved): fixture = Path(__file__).parent / "fixtures" / "atif-journal-v1.jsonl" diff --git a/docs/changelogs/0.8.x.md b/docs/changelogs/0.8.x.md index e4e3325..9b4805e 100644 --- a/docs/changelogs/0.8.x.md +++ b/docs/changelogs/0.8.x.md @@ -12,11 +12,14 @@ All notable changes in the **0.8.x** release series are documented here. - Add a wall-clock time budget to headless runs through `--time-budget-seconds` and the Harbor adapter `time_budget_seconds` kwarg. Tell the model how much time remains each turn, escalate to a stop-and-write - warning inside a reserved finalization window, and stop with a - `time_budget_exhausted` outcome before the harness timeout can kill the - process. Append the reminder to the tail of the conversation so the system - prompt stays fixed and provider prefix caching keeps working. Record the - budget in startup output, Event Journals, and ATIF. + warning inside a reserved finalization window, recheck the deadline before + every tool, and stop with a `time_budget_exhausted` outcome before the + harness timeout can kill the process. Append the reminder to the tail of the + conversation so the system prompt stays fixed and provider prefix caching + keeps working, and record each injected reminder as an `input.injected` + event projected into ATIF. Record the budget in startup output, Event + Journals, and ATIF. New Event Journals use schema v4, with v1-v3 replay still + supported. ### Fixed - Return correctable tool errors for missing or incorrectly typed arguments and diff --git a/docs/dev_docs/README.md b/docs/dev_docs/README.md index 8892f2c..faca3ec 100644 --- a/docs/dev_docs/README.md +++ b/docs/dev_docs/README.md @@ -33,6 +33,9 @@ They complement, rather than replace: - Event Journal Protocol v2 (response truncation): [English](en/event-journal-protocol-v2.md) | [Chinese](zh-CN/event-journal-protocol-v2.md) -- Event Journal Protocol v3 (current; failed model attempts): +- Event Journal Protocol v3 (failed model attempts): [English](en/event-journal-protocol-v3.md) | [Chinese](zh-CN/event-journal-protocol-v3.md) +- Event Journal Protocol v4 (current; time budgets and injected input): + [English](en/event-journal-protocol-v4.md) | + [Chinese](zh-CN/event-journal-protocol-v4.md) diff --git a/docs/dev_docs/en/event-journal-protocol-v3.md b/docs/dev_docs/en/event-journal-protocol-v3.md index edeeffc..689f810 100644 --- a/docs/dev_docs/en/event-journal-protocol-v3.md +++ b/docs/dev_docs/en/event-journal-protocol-v3.md @@ -4,10 +4,12 @@ > [`../zh-CN/event-journal-protocol-v3.md`](../zh-CN/event-journal-protocol-v3.md). > Do not edit by hand. -The current writer emits `schema_version = 3`. Readers and the ATIF projector -continue to accept v1 and v2. Public trajectories remain ATIF-v1.7. -All contracts from [v2](event-journal-protocol-v2.md) still apply except for the -additional event and projection behavior described here. +v3 uses `schema_version = 3`. The current writer has moved to +[v4](event-journal-protocol-v4.md); this document preserves the v3 protocol. +v3 readers and the ATIF projector continue to accept v1 and v2. Public +trajectories remain ATIF-v1.7. All contracts from +[v2](event-journal-protocol-v2.md) still apply except for the additional event +and projection behavior described here. ## Failed model attempts diff --git a/docs/dev_docs/en/event-journal-protocol-v4.md b/docs/dev_docs/en/event-journal-protocol-v4.md new file mode 100644 index 0000000..31c89db --- /dev/null +++ b/docs/dev_docs/en/event-journal-protocol-v4.md @@ -0,0 +1,91 @@ +# Event Journal Implementation Protocol v4 + +> Generated from the Chinese source +> [`../zh-CN/event-journal-protocol-v4.md`](../zh-CN/event-journal-protocol-v4.md). +> Do not edit by hand. + +The current writer emits `schema_version = 4` for all runs. Readers and the +ATIF projector continue to accept v1, v2, and v3. Public trajectories remain +ATIF-v1.7. All contracts from [v3](event-journal-protocol-v3.md) still apply +except for the changes below. + +## Time budget outcome + +`run.completed.payload.outcome` adds `time_budget_exhausted`: the core observed +an exhausted wall-clock budget before starting another model or tool call. +The meanings of `completed`, `max_turns_exhausted`, and `response_truncated` +remain unchanged. + +A headless run with a time budget computes its deadline using a monotonic +clock. It checks before each model attempt and before executing every tool +in a reply. After one tool consumes the remaining budget, subsequent tools +do not start. An in-flight model or tool call is not interrupted and may +outlast the deadline. Existing precedence for response truncation and the +turn budget is unchanged; a complete final reply needing no further work +can still end normally. + +The optional `run.started.time_budget_seconds` field records the configured +positive integer number of seconds, or null when no budget is configured. +Older journals may omit this field. ATIF preserves it in +`agent.extra.time_budget_seconds`. + +Budget exhaustion finalizes normally with `run.completed` and headless exit +code `0`; cost reconciliation and trajectory writing still run. ATIF records +`extra.terminal.status = "completed"` and +`extra.terminal.outcome = "time_budget_exhausted"`. Consumers must inspect +the outcome rather than infer task completion from status alone. + +`model.completed` retains all tools requested by the reply. Only tools that +actually execute have `tool.started`, `tool.completed`, and corresponding +ATIF observations. Tools skipped because of the deadline do not produce +fabricated execution events or results. + +## Injected model input + +The new Native Event `input.injected` records runtime text appended to a +user-role message, separately from the original `user.message`. Its required +payload fields are: + +| Field | Type | Meaning | +| --- | --- | --- | +| `model_call_id` | nonempty string | The upcoming model attempt that will use this input, matching the immediately following `model.started`. | +| `content` | string | The complete text appended this time, not a copy of the whole conversation or tool results. | +| `reason` | nonempty string | Reason for injection; time budget reminders use `time_budget`. | +| `source_timestamp` | RFC 3339 UTC or null | Time of injection. | + +With a time budget configured, every model attempt, including retries, appends +a reminder to the most recent user message and records this event before +`model.started`. The first reminder follows the task text, subsequent reminders +follow tool results, and retries append to the same message. The system prompt +stays unchanged. Each addition is recorded separately, including the final +warning to stop investigating and write the output file. Original user input +and tool results are not rewritten in the Journal. + +ATIF creates a `source = "user"` step for each `input.injected` in Journal +order, with the reminder text in `message`. Here, user denotes the role in the +model request; `extra.injected = true` identifies runtime-generated input. +`extra.model_call_id` and `extra.reason` preserve correlation and purpose. The +step precedes the corresponding model attempt, is not merged into tool +observations, and does not add to `llm_call_count`. + +`content` follows the Journal string persistence limit. When truncated, its +metadata is projected into `extra.journal_truncation`. Correlation identifiers +and `reason` are exempt from string truncation. + +## Compatibility and validation + +Outcomes and event types are closed enums; new values require a schema +increment. v1/v2/v3 records containing `time_budget_exhausted` or +`input.injected` must be rejected, and old readers reject v4. Historical +journals are not rewritten. Existing version boundaries remain: +`response_truncated` is valid from v2 onward, and `model.failed` from v3 onward. + +- [`test_time_budget.py`](../../../tests/test_time_budget.py) covers deadline + checks before each tool and Journal/ATIF persistence of initial reminders, + reminders after tool results, and the final warning. +- [`test_truncation.py`](../../../tests/test_truncation.py) covers outcome + version boundaries; [`test_event_journal.py`](../../../tests/test_event_journal.py) + covers the injected event's version boundary. +- [Harbor compatibility tests](../../../benchmarks/harbor/tests/test_atif_compatibility.py) + validate the v4 projection, including reminders and the budget outcome, + against the pinned official ATIF validator. diff --git a/docs/dev_docs/zh-CN/event-journal-protocol-v3.md b/docs/dev_docs/zh-CN/event-journal-protocol-v3.md index c91718f..1621402 100644 --- a/docs/dev_docs/zh-CN/event-journal-protocol-v3.md +++ b/docs/dev_docs/zh-CN/event-journal-protocol-v3.md @@ -3,8 +3,10 @@ > 本文件为中文源;英文版本 > [`../en/event-journal-protocol-v3.md`](../en/event-journal-protocol-v3.md) 由其生成。 -当前 writer 写入 `schema_version = 3`。Reader 和 ATIF projector 继续兼容 -v1、v2;公开轨迹仍为 ATIF-v1.7。除下述新增事件和投影行为外, +v3 使用 `schema_version = 3`。当前 writer 已升级到 +[v4](event-journal-protocol-v4.md),本文保留 v3 的协议定义。 +v3 reader 和 ATIF projector 继续兼容 v1、v2;公开轨迹仍为 ATIF-v1.7。 +除下述新增事件和投影行为外, [v2](event-journal-protocol-v2.md) 的其他契约保持有效。 ## 模型尝试失败 diff --git a/docs/dev_docs/zh-CN/event-journal-protocol-v4.md b/docs/dev_docs/zh-CN/event-journal-protocol-v4.md new file mode 100644 index 0000000..ad8c550 --- /dev/null +++ b/docs/dev_docs/zh-CN/event-journal-protocol-v4.md @@ -0,0 +1,73 @@ +# Event Journal 实现协议 v4 + +> 本文件为中文源;英文版本 +> [`../en/event-journal-protocol-v4.md`](../en/event-journal-protocol-v4.md) 由其生成。 + +当前 writer 对所有 run 写入 `schema_version = 4`。Reader 和 ATIF projector +继续兼容 v1、v2、v3;公开轨迹仍为 ATIF-v1.7。除以下变化外, +[v3](event-journal-protocol-v3.md) 的契约保持有效。 + +## 时间预算终态 + +`run.completed.payload.outcome` 新增 `time_budget_exhausted`,表示 core 在启动 +下一次模型调用或工具调用前发现 wall-clock 预算已耗尽。原有 `completed`、 +`max_turns_exhausted`、`response_truncated` 的含义保持不变。 + +配置了时间预算的 headless run 使用单调时钟计算截止时间。检查发生在每次模型 +尝试之前,以及同一回复中每一个工具执行之前。一个工具耗尽剩余预算后,后续工具 +不会启动。正在进行的模型或工具调用不会因此被中止,仍可能超过截止时间。 +回复截断和轮数预算的既有优先级保持不变;无需继续工作的完整最终回复仍可正常结束。 + +`run.started` 的可选字段 `time_budget_seconds` 记录配置的正整数秒数;未配置时 +为 null。旧 Journal 可以不包含该字段。ATIF 将其保留在 +`agent.extra.time_budget_seconds`。 + +预算耗尽以 `run.completed` 正常收尾,headless 退出码为 `0`,费用补查和轨迹写入 +仍会执行。ATIF 的 `extra.terminal.status` 为 `completed`, +`extra.terminal.outcome` 为 `time_budget_exhausted`。消费者必须检查 outcome, +不能仅凭 status 推断任务完成。 + +`model.completed` 保留回复请求的全部工具调用;只有实际执行的工具才有 +`tool.started`、`tool.completed` 和对应 ATIF observation。因超时跳过的工具 +不产生虚构的执行事件或结果。 + +## 注入的模型输入 + +新增 Native Event `input.injected`,记录运行时追加到用户角色消息中的文本输入, +与原始 `user.message` 区分。必填字段如下: + +| 字段 | 类型 | 含义 | +| --- | --- | --- | +| `model_call_id` | 非空字符串 | 即将使用该输入的模型尝试标识,与紧随其后的 `model.started` 对应。 | +| `content` | 字符串 | 本次追加的完整文本,不是整个对话或工具结果的副本。 | +| `reason` | 非空字符串 | 注入原因;时间预算提醒使用 `time_budget`。 | +| `source_timestamp` | RFC 3339 UTC 或 null | 注入发生时间。 | + +配置时间预算时,每次模型尝试(包括重试)都会将提醒追加到最近的用户消息尾部, +然后在 `model.started` 之前记录该事件。首次提醒追加到任务文本,后续提醒追加到 +工具结果之后;重试时追加到同一个消息尾部。系统提示保持不变。每次追加都单独 +记录,包括要求停止调查并写入输出文件的最终阶段警告。原始用户输入和工具结果 +不会在 Journal 中被改写。 + +ATIF 按 Journal 顺序为每个 `input.injected` 创建一个 `source = "user"` 的 +step,`message` 为该提醒文本。这里的 user 表示模型请求中的消息角色; +`extra.injected = true` 明确标识它来自运行时。`extra.model_call_id` 和 +`extra.reason` 保留关联与原因。该 step 位于对应模型尝试之前,不合并到工具 +observation,也不增加 `llm_call_count`。 + +`content` 沿用 Journal 字符串持久化上限,发生截断时将相关元数据投影到 +`extra.journal_truncation`。关联标识和 `reason` 不受字符串截断影响。 + +## 兼容性与验证 + +outcome 和事件类型都是封闭枚举,新增值需要递增 schema。v1/v2/v3 记录包含 +`time_budget_exhausted` 或 `input.injected` 时必须拒绝;旧 reader 会拒绝 v4。 +历史 Journal 不会被改写。既有版本边界保持不变:`response_truncated` 从 v2 +开始有效,`model.failed` 从 v3 开始有效。 + +- [`test_time_budget.py`](../../../tests/test_time_budget.py) 覆盖逐工具截止时间检查, + 以及初始提醒、工具结果后的提醒和最终阶段警告在 Journal/ATIF 中的持久化。 +- [`test_truncation.py`](../../../tests/test_truncation.py) 覆盖各版本的 outcome 边界; + [`test_event_journal.py`](../../../tests/test_event_journal.py) 覆盖注入事件的版本边界。 +- [Harbor 兼容性测试](../../../benchmarks/harbor/tests/test_atif_compatibility.py) + 使用固定版本的官方 ATIF validator 验证包含提醒和预算终态的 v4 投影。 diff --git a/src/nanopycodeagent/agent.py b/src/nanopycodeagent/agent.py index 82551e1..5315ecc 100644 --- a/src/nanopycodeagent/agent.py +++ b/src/nanopycodeagent/agent.py @@ -100,7 +100,7 @@ def _format_duration(total_seconds: float) -> str: def _time_budget_note( *, turn: int, elapsed: float, budget: float, remaining: float ) -> str: - """The wall-clock reminder appended to the system prompt for one turn.""" + """The wall-clock reminder appended to the conversation for one turn.""" if remaining <= max( _FINALIZATION_RESERVE_FRACTION * budget, _FINALIZATION_RESERVE_SECONDS ): @@ -119,7 +119,13 @@ def _time_budget_note( ) -def _append_budget_note(messages: list[MessageParam], note: str) -> None: +def _append_budget_note( + messages: list[MessageParam], + note: str, + *, + emitter: EventEmitter, + model_call_id: str, +) -> None: """Append a wall-clock reminder to the tail of the conversation. The reminder goes into the most recent user message — the initial task, or @@ -134,6 +140,15 @@ def _append_budget_note(messages: list[MessageParam], note: str) -> None: last["content"] = f"{content}\n\n{note}" else: content.append({"type": "text", "text": note}) + emitter.emit( + "input.injected", + { + "model_call_id": model_call_id, + "content": note, + "reason": "time_budget", + "source_timestamp": utc_now(), + }, + ) # Shared by both system prompts: which tool to reach for is the same question # whoever is asking. @@ -558,6 +573,7 @@ def _run_model_loop( else None ) while True: + model_call_id = f"model-{uuid.uuid4()}" if deadline is not None: remaining = deadline - time.monotonic() if remaining <= 0: @@ -570,11 +586,12 @@ def _run_model_loop( budget=time_budget_seconds, remaining=remaining, ), + emitter=emitter, + model_call_id=model_call_id, ) # A spinner marks the wait for the reply; the first streamed # token replaces it with the reply prefix. A tool-only reply # streams no text, so the prefix is skipped for it entirely. - model_call_id = f"model-{uuid.uuid4()}" emitter.emit( "model.started", { @@ -734,20 +751,22 @@ def _run_model_loop( # Stop before running the tools: their results would only be # useful to a reply this budget can no longer pay for. return "max_turns_exhausted" - if deadline is not None and deadline - time.monotonic() <= 0: - # The reply consumed the remaining budget; its tool results would - # only be useful to a turn this run can no longer afford. - return "time_budget_exhausted" # Every tool_use block needs a matching tool_result in the next # user message, or the API rejects the request. - results = [ - _run_one_tool( - block, emitter, model_call_id, - input_error=input_errors.get(block.id), + results = [] + for block in message.content: + if block.type != "tool_use": + continue + if deadline is not None and time.monotonic() >= deadline: + # The reply or a preceding tool consumed the remaining time. + # Stop the run without starting another tool or model call. + return "time_budget_exhausted" + results.append( + _run_one_tool( + block, emitter, model_call_id, + input_error=input_errors.get(block.id), + ) ) - for block in message.content - if block.type == "tool_use" - ] messages.append({"role": "user", "content": results}) diff --git a/src/nanopycodeagent/atif.py b/src/nanopycodeagent/atif.py index f728dfc..9ae019b 100644 --- a/src/nanopycodeagent/atif.py +++ b/src/nanopycodeagent/atif.py @@ -274,12 +274,17 @@ def project_atif(entries: Sequence[JournalEntry]) -> JsonObject: terminal: JournalEntry | None = None for entry in entries[1:]: payload = entry.payload - if entry.type == "user.message": + if entry.type in {"user.message", "input.injected"}: timestamp, timestamp_source = _timestamp(entry) - extra: JsonObject = { - "message_id": payload["message_id"], - "timestamp_source": timestamp_source, - } + extra: JsonObject = {"timestamp_source": timestamp_source} + if entry.type == "input.injected": + extra.update({ + "injected": True, + "model_call_id": payload["model_call_id"], + "reason": payload["reason"], + }) + else: + extra["message_id"] = payload["message_id"] _add_journal_truncation(extra, entry, "/content") steps.append( { diff --git a/src/nanopycodeagent/event_journal.py b/src/nanopycodeagent/event_journal.py index 56c9013..09a2e81 100644 --- a/src/nanopycodeagent/event_journal.py +++ b/src/nanopycodeagent/event_journal.py @@ -19,8 +19,8 @@ from . import settings -SCHEMA_VERSION = 3 -SUPPORTED_SCHEMA_VERSIONS = frozenset({1, 2, 3}) +SCHEMA_VERSION = 4 +SUPPORTED_SCHEMA_VERSIONS = frozenset({1, 2, 3, 4}) DEFAULT_MAX_STRING_CHARS = 100_000 type RunOutcome = Literal[ @@ -31,6 +31,7 @@ { "run.started", "user.message", + "input.injected", "model.started", "model.output_delta", "model.completed", @@ -60,6 +61,7 @@ "outcome", "producer", "provider_response_id", + "reason", "source_timestamp", "stop_reason", "tool_call_id", @@ -72,6 +74,9 @@ {"mode", "model", "max_turns", "producer", "source_timestamp"} ), "user.message": frozenset({"message_id", "content", "source_timestamp"}), + "input.injected": frozenset( + {"model_call_id", "content", "reason", "source_timestamp"} + ), "model.started": frozenset({"model_call_id", "model", "source_timestamp"}), "model.output_delta": frozenset( {"model_call_id", "delta", "source_timestamp"} @@ -334,6 +339,10 @@ def _validate_native_payload(event_type: str, payload: JsonObject) -> None: or max_turns < 1 ): raise ValueError("run.started.max_turns must be positive or null") + elif event_type == "input.injected": + _require_string(payload, "reason", event_type) + if not isinstance(payload["content"], str): + raise ValueError("input.injected.content must be a string") elif event_type == "model.started": _require_string(payload, "model", event_type) elif event_type == "model.output_delta": @@ -550,8 +559,16 @@ def from_dict(cls, value: Mapping[str, object]) -> JournalEntry: if truncation is not None: _validate_truncation(truncation) event = NativeEvent(event_type, payload) + if schema_version < 4 and event.type == "input.injected": + raise ValueError("input.injected requires Journal Entry schema 4") if schema_version < 3 and event.type == "model.failed": raise ValueError("model.failed requires Journal Entry schema 3") + if ( + schema_version < 4 + and event.type == "run.completed" + and event.payload["outcome"] == "time_budget_exhausted" + ): + raise ValueError("time_budget_exhausted requires Journal Entry schema 4") if ( schema_version == 1 and event.type == "run.completed" diff --git a/tests/test_event_journal.py b/tests/test_event_journal.py index 4b5298a..7fb9fa0 100644 --- a/tests/test_event_journal.py +++ b/tests/test_event_journal.py @@ -54,7 +54,7 @@ def test_journal_entry_wraps_the_native_event_with_ordering_metadata(tmp_path): }, } assert entry.to_dict() == { - "schema_version": 3, + "schema_version": 4, "run_id": "run-123", "seq": 1, "recorded_at": "2026-08-23T08:00:01.420Z", @@ -105,6 +105,28 @@ def test_jsonl_journal_replays_entries_in_append_order(tmp_path): assert path.read_bytes().count(b"\n") == 2 +@pytest.mark.parametrize("schema_version", [1, 2, 3, 4]) +def test_injected_input_requires_v4_schema(schema_version): + record = { + "schema_version": schema_version, + "run_id": "run-1", + "seq": 1, + "recorded_at": "2026-09-27T00:00:00.000Z", + "type": "input.injected", + "payload": { + "model_call_id": "model-1", + "content": "[time budget] Stop investigating now.", + "reason": "time_budget", + "source_timestamp": None, + }, + } + if schema_version < 4: + with pytest.raises(ValueError, match="input.injected requires Journal Entry schema 4"): + JournalEntry.from_dict(record) + else: + assert JournalEntry.from_dict(record).to_dict() == record + + def test_journal_storage_is_restricted_to_the_current_user(tmp_path): directory = tmp_path / "journals" directory.mkdir(mode=0o755) @@ -336,7 +358,7 @@ def test_native_event_contract_rejects_non_json_values(content): ) -@pytest.mark.parametrize("schema_version", [True, 0, 4, "2"]) +@pytest.mark.parametrize("schema_version", [True, 0, 5, "2"]) def test_journal_entry_rejects_unsupported_schema_version(schema_version): with pytest.raises(ValueError, match="unsupported Journal Entry schema"): JournalEntry.from_dict( diff --git a/tests/test_stream_recovery.py b/tests/test_stream_recovery.py index 0f288a9..3292403 100644 --- a/tests/test_stream_recovery.py +++ b/tests/test_stream_recovery.py @@ -208,7 +208,7 @@ def resolve(base_url, generation_id, credential, **kwargs): assert "total_prompt_tokens" not in final -@pytest.mark.parametrize("schema", [1, 2, 3]) +@pytest.mark.parametrize("schema", [1, 2, 3, 4]) def test_failed_attempt_requires_v3_schema(schema): record = { "schema_version": schema, "run_id": "run-1", "seq": 1, diff --git a/tests/test_time_budget.py b/tests/test_time_budget.py index 67660bb..5260432 100644 --- a/tests/test_time_budget.py +++ b/tests/test_time_budget.py @@ -6,6 +6,8 @@ written. """ +import json + import pytest from nanopycodeagent import agent, cli, settings @@ -148,6 +150,106 @@ def test_without_a_budget_the_system_prompt_is_unchanged(monkeypatch): entries = _journal_entries() assert entries[0].payload["time_budget_seconds"] is None assert entries[-1].payload["outcome"] == "completed" + assert not any(entry.type == "input.injected" for entry in entries) + + +@pytest.mark.parametrize("tool_seconds", [1, 2]) +def test_deadline_is_checked_before_each_tool(monkeypatch, tmp_path, tool_seconds): + clock = FakeTime() + monkeypatch.setattr(agent, "time", clock) + executions = [] + + def run_bash(command): + executions.append(command) + clock.advance(tool_seconds) + return "first result", False + + monkeypatch.setattr(agent, "run_bash", run_bash) + messages = FakeMessages([ + FakeStream([ + tool_use_block("call-1", "first"), + text_block("then another command"), + tool_use_block("call-2", "second"), + ], stop_reason="tool_use"), + ]) + patch_client(monkeypatch, FakeClient(messages)) + trajectory_path = tmp_path / "trajectory.json" + + assert agent.run_headless( + "task", time_budget_seconds=1, trajectory_path=trajectory_path + ) == 0 + + assert executions == ["first"] + assert len(messages.calls) == 1 + entries = _journal_entries() + assert entries[-1].payload["outcome"] == "time_budget_exhausted" + tool_events = [entry for entry in entries if entry.type.startswith("tool.")] + assert [(entry.type, entry.payload["tool_call_id"]) for entry in tool_events] == [ + ("tool.started", "call-1"), ("tool.completed", "call-1"), + ] + trajectory = json.loads(trajectory_path.read_text()) + assert trajectory["extra"]["terminal"]["outcome"] == "time_budget_exhausted" + model_step = next(step for step in trajectory["steps"] if step["source"] == "agent") + assert [call["tool_call_id"] for call in model_step["tool_calls"]] == ["call-1", "call-2"] + assert [result["source_call_id"] for result in model_step["observation"]["results"]] == ["call-1"] + + +@pytest.mark.parametrize("structured_task", [False, True]) +def test_budget_reminders_survive_journal_and_atif(monkeypatch, tmp_path, structured_task): + clock = FakeTime() + monkeypatch.setattr(agent, "time", clock) + monkeypatch.setattr(agent, "run_bash", _fake_bash([])) + messages = AdvancingFakeMessages([ + FakeStream([tool_use_block("call-1", "first")], stop_reason="tool_use"), + FakeStream([tool_use_block("call-2", "second")], stop_reason="tool_use"), + FakeStream([text_block("done")]), + ], clock, 450) + task = [{"type": "text", "text": "task"}] if structured_task else "task" + trajectory_path = tmp_path / "trajectory.json" + + assert agent._run_exchange( + FakeClient(messages), "test-model", [{"role": "user", "content": task}], + agent.HEADLESS_SYSTEM_PROMPT, max_turns=10, time_budget_seconds=1000, + trajectory_path=trajectory_path, + ) == "completed" + + notes = [] + for call in messages.calls: + content = call[-1]["content"] + notes.append(content.split("\n\n")[-1] if isinstance(content, str) else content[-1]["text"]) + assert len(notes) == 3 + assert "Elapsed 0:00" in notes[0] + assert "Elapsed 7:30" in notes[1] + assert "Stop investigating now" in notes[2] + assert "write your best answer to it" in notes[2] + + entries = _journal_entries() + user_entry = next(entry for entry in entries if entry.type == "user.message") + assert user_entry.payload["content"] == ( + [{"type": "text", "text": "task"}] if structured_task else "task" + ) + reminders = [entry for entry in entries if entry.type == "input.injected"] + assert [entry.payload["content"] for entry in reminders] == notes + for entry in reminders: + following = entries[entries.index(entry) + 1] + assert following.type == "model.started" + assert entry.payload["model_call_id"] == following.payload["model_call_id"] + assert entry.payload["reason"] == "time_budget" + + trajectory = json.loads(trajectory_path.read_text()) + steps = trajectory["steps"] + assert [step["source"] for step in steps] == [ + "user", "user", "agent", "user", "agent", "user", "agent", + ] + injected = [step for step in steps if step.get("extra", {}).get("injected")] + assert [step["message"] for step in injected] == notes + for step, entry in zip(injected, reminders, strict=True): + assert step["extra"]["model_call_id"] == entry.payload["model_call_id"] + assert step["extra"]["reason"] == "time_budget" + for index, tool_id in ((2, "call-1"), (4, "call-2")): + result = steps[index]["observation"]["results"][0] + assert result["source_call_id"] == tool_id + assert result["content"] == "ok" @pytest.mark.parametrize("value", ["0", "-3"]) diff --git a/tests/test_truncation.py b/tests/test_truncation.py index 4b96707..96451f8 100644 --- a/tests/test_truncation.py +++ b/tests/test_truncation.py @@ -75,7 +75,7 @@ def resolve(base_url, generation_id, credential, **kwargs): assert reconciled == ["gen-truncated"] entries = _journal_entries() - assert all(entry.schema_version == 3 for entry in entries) + assert all(entry.schema_version == 4 for entry in entries) assert entries[-1].type == "run.completed" assert entries[-1].payload["outcome"] == "response_truncated" completed = next(entry for entry in entries if entry.type == "model.completed") @@ -182,16 +182,19 @@ def respond(request): assert "observation" not in trajectory["steps"][1] -@pytest.mark.parametrize("schema_version", [1, 2, 3]) -@pytest.mark.parametrize("outcome", ["completed", "max_turns_exhausted", "response_truncated"]) +@pytest.mark.parametrize("schema_version", [1, 2, 3, 4]) +@pytest.mark.parametrize("outcome", [ + "completed", "max_turns_exhausted", "response_truncated", "time_budget_exhausted", +]) def test_journal_outcome_versions(schema_version, outcome): record = { "schema_version": schema_version, "run_id": "run-1", "seq": 1, "recorded_at": "2026-09-07T00:00:00.000Z", "type": "run.completed", "payload": {"outcome": outcome, "duration_ms": 1, "source_timestamp": None}, } - if schema_version == 1 and outcome == "response_truncated": - with pytest.raises(ValueError, match="requires Journal Entry schema 2"): + minimum_version = {"response_truncated": 2, "time_budget_exhausted": 4}.get(outcome, 1) + if schema_version < minimum_version: + with pytest.raises(ValueError, match=f"requires Journal Entry schema {minimum_version}"): JournalEntry.from_dict(record) else: assert JournalEntry.from_dict(record).to_dict() == record From c4c35dfbce5f9794d31eafef02d04bc7886e7687 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Sun, 27 Sep 2026 23:01:34 +0800 Subject: [PATCH 07/12] docs: record the tail-injection time-budget experiment Experiment 3 moves the budget reminder to the conversation tail. Both tasks passed and the prefix cache recovered (input tokens 2.92M -> 0.109M, cost $0.856 -> $0.187), but mteb-leaderboard ran to the 100-turn cap instead of stopping on its own. --- .../tb21-exp3-timebudget-tail-20260926.json | 115 ++++++++++++++++++ docs/dev_notes/en/0.8.x.md | 68 +++++++++++ docs/dev_notes/zh-CN/0.8.x.md | 58 +++++++++ 3 files changed, 241 insertions(+) create mode 100644 benchmarks/harbor/results/tb21-exp3-timebudget-tail-20260926.json diff --git a/benchmarks/harbor/results/tb21-exp3-timebudget-tail-20260926.json b/benchmarks/harbor/results/tb21-exp3-timebudget-tail-20260926.json new file mode 100644 index 0000000..a77e5ab --- /dev/null +++ b/benchmarks/harbor/results/tb21-exp3-timebudget-tail-20260926.json @@ -0,0 +1,115 @@ +{ + "schema_version": 1, + "generated_at": "2026-09-27T15:00:55.652126+00:00", + "purpose": "Experiment 3: same as Experiment 2 but the budget reminder is appended to the conversation tail instead of the system prompt, to keep provider prefix caching working. Two tasks, one trial each.", + "agent_ref": "83e5c064d4c6e421fc08a90bbd84d2d23b6bb3cf", + "agent_version": "0.8.1.dev37+g83e5c064d", + "model": "openrouter/deepseek/deepseek-v4.1-flash", + "dataset": "terminal-bench/terminal-bench-2-1", + "dataset_ref": "sha256:7d7bdc1cbedad549fc1140404bd4dc45e5fd0ea7c4186773687d177ad3a0699a", + "selection": "The two tasks that still scored reward 0 in tb21-v41flash-6task-20260925 (caffe-cifar-10, mteb-leaderboard) with their pinned task refs.", + "configuration": { + "max_turns": 100, + "time_budget_seconds": 3420, + "previous_max_turns": 50, + "max_tokens": 65536, + "project_default_max_tokens": 32768, + "agent_execution_limit_seconds": 3480, + "finalization_grace_seconds": 120, + "harbor_outer_limit_seconds": null, + "agent_setup_limit_seconds": null, + "timeout_compliance": "Leaderboard-compliant: no Harbor agent/verifier timeout override and no setup timeout override; the effective agent timeout is the task-native 3600s. The agent self-limits at 3480s inside it.", + "maximum_simultaneous_trials": 2, + "automatic_harbor_retries": 0, + "attempts_per_task": 1, + "verifier_limits": "Native task deadlines", + "dependency_constraints": [ + "anthropic==1.5.0", + "httpx==0.28.1", + "httpx2==2.12.0", + "pydantic==2.13.5" + ] + }, + "baseline": { + "job": "tb21-v41flash-6task-20260925", + "max_turns": 50, + "note": "Both re-run tasks scored reward 0 there by exhausting 50 turns (max_turns_exhausted)." + }, + "totals": { + "planned": 2, + "finished": 2, + "passed": 2, + "scored_zero": 0, + "unscored": 0, + "model_attempts": 159, + "input_tokens": 108646, + "cache_tokens": 8410880, + "output_tokens": 86787, + "cost_usd": 0.18695358 + }, + "trials": [ + { + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "reward": 1.0, + "outcome": "completed", + "duration_seconds": 3333.4, + "model_replies": 59, + "tool_calls": 72, + "input_tokens": 24842, + "cache_tokens": 1520384, + "output_tokens": 26593, + "cost_usd": 0.048236604, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "started_at": "2026-09-27T13:47:58.276Z" + }, + { + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "reward": 1.0, + "outcome": "max_turns_exhausted", + "duration_seconds": 1364.9, + "model_replies": 100, + "tool_calls": 112, + "input_tokens": 83804, + "cache_tokens": 6890496, + "output_tokens": 60194, + "cost_usd": 0.138716976, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "started_at": "2026-09-27T13:47:47.144Z" + } + ], + "comparison_limits": "Single attempt per task; routing, sampling, cache and provider backend are uncontrolled, so this is a targeted diagnostic, not a controlled estimate of pass-rate improvement.", + "artifacts": { + "raw_trials": "jobs/tb21-exp3-timebudget-tail-20260926", + "record": "jobs/tb21-exp3-timebudget-tail-20260926-record", + "note": "Raw trials and record live in Git-ignored jobs/ and are not uploaded." + }, + "provider_routing": { + "by_provider": { + "CoreWeave": { + "replies": 1, + "input_tokens": 1575, + "cache_read_tokens": 0, + "output_tokens": 168 + }, + "Parasail": { + "replies": 157, + "input_tokens": 105565, + "cache_read_tokens": 8410880, + "output_tokens": 86417 + }, + "Together": { + "replies": 1, + "input_tokens": 1506, + "cache_read_tokens": 0, + "output_tokens": 202 + } + }, + "note": "Routing is uncontrolled. The budget reminder is appended to the conversation tail, so the system prompt stays constant. Parasail handled 157 of 159 replies with ~53572 cache reads per reply; in Experiment 2 the reminder changed the system prompt every turn and Parasail reported 0 cache reads." + } +} diff --git a/docs/dev_notes/en/0.8.x.md b/docs/dev_notes/en/0.8.x.md index c8e02cf..e57e83b 100644 --- a/docs/dev_notes/en/0.8.x.md +++ b/docs/dev_notes/en/0.8.x.md @@ -1294,3 +1294,71 @@ in Git-ignored `jobs/` and were not uploaded. Version control stores the structured result [`tb21-exp2-timebudget-20260926.json`](../../../benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json) and this summary. + +### Experiment 3: moving the budget reminder to the conversation tail + +**Run on 2026-09-27.** Experiment 2 found that rewriting the system prompt every +turn defeated the provider's prefix cache. The only change in this run is that the +**system prompt stays fixed and the budget reminder is appended to the tail of the +conversation** (after the task text on the first turn, after the tool results +afterwards); everything else matches Experiment 2 exactly (max_turns 100, a 3420s +budget, no Harbor timeout overrides). + +| Item | This run | +| --- | --- | +| Job | `tb21-exp3-timebudget-tail-20260926` | +| Agent source | branch `experiment/max-turns-and-time-budget`, commit `83e5c064d`; wheel `0.8.1.dev37+g83e5c064d` | +| Everything else | same as Experiment 2: 3420s budget, at most 100 replies per task, no timeout override | + +**Result: both tasks passed.** + +| Task | Reward | Terminal state | Model replies | Duration | Cost (USD) | +| --- | ---: | --- | ---: | ---: | ---: | +| `caffe-cifar-10` | 1 | `completed` | 59 | 3333 s | 0.048236604 | +| `mteb-leaderboard` | 1 | `max_turns_exhausted` | 100 | 1365 s | 0.138716976 | + +**The caching fix is verified.** The two runs side by side (two tasks each): + +| Metric | Experiment 2 (system injection) | Experiment 3 (tail injection) | +| --- | ---: | ---: | +| Input tokens | 2,919,588 | **108,646** | +| Cache tokens | 1,408 | **8,410,880** | +| Total cost (USD) | 0.856 | **0.187** | + +The clearest evidence is the same-provider before/after: in Experiment 2 Parasail's +29 calls reported zero cache reads; in Experiment 3 Parasail served 157 of 159 +replies with about **53,572** cached tokens each. Going from zero to fifty-odd +thousand on the same provider is consistent with "fixed system prompt plus prefix +cache". + +**But convergence got weaker.** Experiment 2's `mteb-leaderboard` stopped on its +own at 50 replies (`completed`); in Experiment 3 it wrote the correct +`/app/result.txt` (reward still 1) but kept working to the 100-reply cap +(`max_turns_exhausted`) and never produced a closing summary. Either the tail +reminder is less salient than a system-prompt instruction, or this is run-to-run +noise — with one trial per run, this run cannot tell the two apart. + +**Code review and fixes.** An independent codex `/review` (against `main`) audited +this branch's implementation and raised three issues, now fixed in the next +section: + +- **[P1] The deadline was checked once per batch**: when a reply contains several + tool calls, the rest still run after the first exhausts the budget. Now the + deadline is **rechecked before every tool**. +- **[P2] Injected reminders were not persisted**: the budget reminder is + behavior-changing input but appeared in neither the Event Journal nor ATIF. A new + `input.injected` event and matching ATIF step were added. +- **[P2] The new outcome was not versioned**: a run ending in + `time_budget_exhausted` still claimed schema v3, which old v3 readers reject, + while the updated reader wrongly accepted it in historical v1/v2 records. + Following the `response_truncated` precedent, the **schema is bumped to v4**. + +**Evidence boundary.** Still one trial per task, with routing uncontrolled +(Experiment 3 sent 157 of 159 calls to Parasail). The cache difference has a clear +direction, but "weaker convergence" is only a signal. + +**Artifacts.** Raw trials are under `jobs/tb21-exp3-timebudget-tail-20260926/`, with +the experiment record under `jobs/tb21-exp3-timebudget-tail-20260926-record/`. +Version control stores the structured result +[`tb21-exp3-timebudget-tail-20260926.json`](../../../benchmarks/harbor/results/tb21-exp3-timebudget-tail-20260926.json) +and this summary. diff --git a/docs/dev_notes/zh-CN/0.8.x.md b/docs/dev_notes/zh-CN/0.8.x.md index b70bb5e..31727e1 100644 --- a/docs/dev_notes/zh-CN/0.8.x.md +++ b/docs/dev_notes/zh-CN/0.8.x.md @@ -1681,3 +1681,61 @@ tokens 从 4.29M 掉到 1.4K,总费用从 $0.094 涨到 $0.856。按 provider 版本管理保存结构化结果 [`tb21-exp2-timebudget-20260926.json`](../../../benchmarks/harbor/results/tb21-exp2-timebudget-20260926.json) 与本节汇总。 + +### 实验三:把预算提示移到对话尾部(修复缓存) + +**2026-09-27 实跑。** 实验二发现逐轮改写 system prompt 会让 provider 的前缀缓存 +失效。本轮唯一的改动是:**system prompt 保持固定,把预算提示追加到对话尾部** +(第 1 轮跟在任务文本后,之后跟在 tool 结果后),其余与实验二完全一致(max_turns +100、预算 3420 秒、无任何 Harbor timeout 覆盖)。 + +| 项目 | 本轮配置 | +| --- | --- | +| Job | `tb21-exp3-timebudget-tail-20260926` | +| Agent 源码 | 分支 `experiment/max-turns-and-time-budget`,提交 `83e5c064d`;wheel `0.8.1.dev37+g83e5c064d` | +| 其余 | 与实验二相同:预算 3420 秒、每题最多 100 次回复、无 timeout 覆盖 | + +**结果:2 题全部通过。** + +| 题目 | Reward | 终态 | 模型回复 | 时长 | 费用(USD) | +| --- | ---: | --- | ---: | ---: | ---: | +| `caffe-cifar-10` | 1 | `completed` | 59 | 3333 s | 0.048236604 | +| `mteb-leaderboard` | 1 | `max_turns_exhausted` | 100 | 1365 s | 0.138716976 | + +**缓存修复已验证。** 两轮对比(同为 2 题): + +| 指标 | 实验二(system 注入) | 实验三(尾部注入) | +| --- | ---: | ---: | +| input tokens | 2,919,588 | **108,646** | +| cache tokens | 1,408 | **8,410,880** | +| 总费用(USD) | 0.856 | **0.187** | + +更直接的证据来自同一 provider 的前后对比:实验二里 Parasail 的 29 次调用命中 0 个 +缓存读取;实验三里 Parasail 承担了 159 次中的 157 次,平均每次命中约 **53,572** 个 +缓存 tokens。同一 provider 上从 0 恢复到 5 万多,与"system 固定 + 前缀缓存生效" +一致。 + +**但收敛行为变弱了。** 实验二的 `mteb-leaderboard` 在 50 轮主动 `completed`; +实验三它写对了 `/app/result.txt`(reward 仍为 1),却一路做到第 100 轮被 +`max_turns_exhausted` 截断,全程没有输出任何收尾文本。两种可能:尾部提示嵌在 +tool 结果里,显著性不如 system prompt;也可能只是单次运行波动。无论哪种,**这一 +轮仍只跑 1 次,不能区分二者**。 + +**代码 review 与修复。** 借独立的 codex `/review`(按 main 分支对比)复查了本分支 +的实现,它提出 3 个问题并已修复(见下节): + +- **[P1] 只在批次前检查一次截止时间**:一次回复里有多个工具调用时,第一个工具 + 耗尽预算后其余仍会执行。改为**执行每个工具前重新检查**。 +- **[P2] 注入的提示没有进入轨迹**:预算提示是改变行为的外部输入,却既不在 Event + Journal 也不在 ATIF 里。新增 `input.injected` 事件和对应的 ATIF step。 +- **[P2] 新终态没有版本化**:`time_budget_exhausted` 让新写入仍声称 schema v3, + 而旧 v3 reader 会拒绝它,更新后的 reader 又会错误接受历史 v1/v2 记录。按 + `response_truncated` 的先例**把 Schema 升到 v4**。 + +**证据边界。** 每题仍只跑 1 次,路由未受控(实验三 157/159 次落在 Parasail)。 +缓存差异的方向明确,但"收敛变弱"只能算信号。 + +**产物位置。** 原始 trial 在 `jobs/tb21-exp3-timebudget-tail-20260926/`,实验记录在 +`jobs/tb21-exp3-timebudget-tail-20260926-record/`。版本管理保存结构化结果 +[`tb21-exp3-timebudget-tail-20260926.json`](../../../benchmarks/harbor/results/tb21-exp3-timebudget-tail-20260926.json) +与本节汇总。 From 847cc98956562b484897780ba662ee9f42df2cd3 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Tue, 29 Sep 2026 21:24:08 +0800 Subject: [PATCH 08/12] feat(agent): finalize within time and turn budgets --- .../harbor/reports/dual-budget-20260929.md | 70 +++++++++ src/nanopycodeagent/agent.py | 143 +++++++++++++----- src/nanopycodeagent/bash_tool.py | 47 ++++-- src/nanopycodeagent/cli.py | 13 +- src/nanopycodeagent/deadline.py | 39 +++++ tests/helpers.py | 3 +- tests/test_agent_events.py | 13 +- tests/test_bash_tool.py | 24 +++ tests/test_cli.py | 17 ++- tests/test_stream_recovery.py | 6 +- tests/test_time_budget.py | 132 ++++++++++++++-- tests/test_tool_validation.py | 8 +- tests/test_truncation.py | 6 +- 13 files changed, 434 insertions(+), 87 deletions(-) create mode 100644 benchmarks/harbor/reports/dual-budget-20260929.md create mode 100644 src/nanopycodeagent/deadline.py diff --git a/benchmarks/harbor/reports/dual-budget-20260929.md b/benchmarks/harbor/reports/dual-budget-20260929.md new file mode 100644 index 0000000..2f8a903 --- /dev/null +++ b/benchmarks/harbor/reports/dual-budget-20260929.md @@ -0,0 +1,70 @@ +# Dual-budget finalization with a fixed provider + +## Implementation + +Headless runs receive append-only reminders of their remaining model replies +and, when configured, wall-clock time. The system prompt stays fixed. Either +10 remaining replies (including the next reply) or the existing time reserve +(`max(180 seconds, 15% of the task budget)`) triggers finalization guidance: +save required deliverables, perform necessary checks, then summarize and stop. +The final allowed reply cannot execute tools; the reminder states this limit. +The ordinary CLI and Harbor defaults remain 50 replies and 32768 output tokens. + +Configured wall-clock budgets interrupt blocking model streams and tool work +using a POSIX main-thread timer. Socket and bash timeouts also use the remaining +budget. Bash interruptions terminate the command's process group; normal exits +preserve background services. Stream retries must fit inside the remaining +budget. Cost reconciliation has a separate shared 30-second cap, after which +unresolved costs remain pending and the native Journal/ATIF can be finalized. +Callers with an existing real-time alarm, non-main threads, and platforms +without `setitimer` receive an explicit error for time-budgeted runs. + +## Registered comparison + +- Model: `deepseek/deepseek-v4.1-flash`. +- Provider endpoint: `parasail/fp8`, with both `only` and `order` restricted to + that endpoint and `allow_fallbacks=false`. Pinning is applied by the experiment + wrapper to both arms, not through a change to account-wide routing settings. +- Control: the source/wheel used by experiment 3 (`83e5c064d`). +- Treatment: the committed dual-budget implementation in this branch. +- Tasks: the same pinned `caffe-cifar-10` and `mteb-leaderboard` task refs. +- Three attempts per task per arm, in separate preserved jobs, alternating arm + order across pairs. Two simultaneous trials, 100 replies, 65536 output tokens, + 3420 seconds for task work, no whole-task retries. +- Native Harbor task/setup/verifier timeouts are preserved. The supervisor + remains at 3480 seconds with 120 seconds of finalization grace. +- Dependencies and cached image digests match experiment 3. No fault injection. +- Verify actual provider receipts for all model requests, not just the routing + configuration. Preserve incomplete receipts as unknown rather than treating + missing costs as zero; post-run reconciliation may supplement benchmark + accounting without modifying the agent's native Journal or trajectory. + +The initial provider preflight returned a tool call and a generation receipt +identifying Parasail. OpenRouter documents `provider` on its +[Messages endpoint](https://openrouter.ai/docs/api/api-reference/anthropic-messages/create-messages) +and endpoint-specific restrictions in its +[routing reference](https://openrouter.ai/docs/guides/routing/provider-selection). + +## Gate for pilot20 + +The treatment must score reward 1 and end as native `completed` on all six +targeted trials, with no supervisor deadline, forced kill, or reconstructed +trajectory. Provider auditing must succeed. Compare per-task cost and duration +against the repeated control before proceeding; investigate a material +regression rather than selecting the best run. If the gate passes, run the +original pinned pilot20 selection once with the same model and provider, 100 +replies, and task-specific budgets inside each native timeout. Historical +pilot20 used another model/routing configuration and is not a controlled +estimate of this implementation's effect. + +These small repeated samples can reveal regressions but do not establish a +general optimal default. Provider load, cache state, and sampling remain +sources of variation despite a fixed endpoint. + +## Local validation before live trials + +The pinned benchmark environment passes 351 tests across the main package and +Harbor adapter. Tests cover early turn-based finalization, stable system +prompts, reminder journaling, stalled streams, real long-running commands, +child-process termination, preservation of background services after normal +exit, bounded retries, bounded cost reconciliation, and ATIF compatibility. diff --git a/src/nanopycodeagent/agent.py b/src/nanopycodeagent/agent.py index 5315ecc..cb4ee32 100644 --- a/src/nanopycodeagent/agent.py +++ b/src/nanopycodeagent/agent.py @@ -47,6 +47,7 @@ usage_cost, ) from .edit_tool import edit_preview, run_edit +from .deadline import DeadlineExceeded, check_deadline_support, wall_clock_limit from .event_journal import ( EventEmitter, EventJournal, @@ -90,6 +91,8 @@ # reserved so the model still has room to write the task's output file. _FINALIZATION_RESERVE_SECONDS = 180 _FINALIZATION_RESERVE_FRACTION = 0.15 +_FINALIZATION_RESERVE_TURNS = 10 +_COST_RECONCILIATION_SECONDS = 30 def _format_duration(total_seconds: float) -> str: @@ -98,25 +101,41 @@ def _format_duration(total_seconds: float) -> str: def _time_budget_note( - *, turn: int, elapsed: float, budget: float, remaining: float + *, turn: int, elapsed: float, budget: float | None, remaining: float | None, + max_turns: int | None = None, ) -> str: - """The wall-clock reminder appended to the conversation for one turn.""" - if remaining <= max( + """Tell the model about both limits before either prevents finalization.""" + replies_left = None if max_turns is None else max_turns - turn + 1 + turn_note = f"Turn {turn}. " if max_turns is None else ( + f"Turn {turn} of {max_turns}; {replies_left} replies remaining including " + "this one. The last reply must be a final summary: its tool calls " + "will not execute. " + ) + time_low = remaining is not None and remaining <= max( _FINALIZATION_RESERVE_FRACTION * budget, _FINALIZATION_RESERVE_SECONDS - ): - return ( - f"[time budget] Turn {turn}. Only {_format_duration(remaining)} of " - f"{_format_duration(budget)} left. Stop investigating now. If the " - "task names an output file, write your best answer to it, verify " - "it exists, then reply with a short summary and no further tool " - "calls." - ) - return ( - f"[time budget] Turn {turn}. Elapsed {_format_duration(elapsed)} of " - f"{_format_duration(budget)}; {_format_duration(remaining)} remaining. " - "This is a hard wall-clock limit: the run stops when it expires. Keep " - "any required output file up to date and reserve time to finish." ) + note = "[runtime budget] " + turn_note + if remaining is not None: + note += ( + f"Only {_format_duration(remaining)} of {_format_duration(budget)} left. " + if time_low else + f"Elapsed {_format_duration(elapsed)} of {_format_duration(budget)}; " + f"{_format_duration(remaining)} remaining. " + ) + if time_low or (replies_left is not None and replies_left <= _FINALIZATION_RESERVE_TURNS): + note += ( + "Finalize now. Stop investigating new approaches. Complete and save " + "the required deliverables, perform only the necessary checks, then " + "reply with a short summary and no further tool calls. If incomplete, " + "save useful progress and state the remaining limitation honestly." + ) + else: + note += ( + "Keep required deliverables up to date. Once the requirements are " + "satisfied and checked, finish immediately; unused budget is not " + "a reason to continue investigating." + ) + return note def _append_budget_note( @@ -125,6 +144,7 @@ def _append_budget_note( *, emitter: EventEmitter, model_call_id: str, + reason: str = "time_budget", ) -> None: """Append a wall-clock reminder to the tail of the conversation. @@ -145,7 +165,7 @@ def _append_budget_note( { "model_call_id": model_call_id, "content": note, - "reason": "time_budget", + "reason": reason, "source_timestamp": utc_now(), }, ) @@ -175,7 +195,10 @@ def _append_budget_note( "out. Work the task through to the end, then check the result with the " "tools instead of assuming it worked. When it is done, answer with a " "short summary and no further tool calls: that reply is what ends the " - "run. " + "run. Runtime budget reminders report remaining time and model replies; " + "when either is running low, prioritize saving the required deliverables, " + "necessary verification, and a final summary. Do not start optional work " + "after the requirements are satisfied. " ) + _TOOL_GUIDANCE def _json_value(value: object) -> JsonValue: @@ -332,6 +355,7 @@ def _run_one_tool( model_call_id: str, *, input_error: str | None = None, + remaining_seconds: float | None = None, ) -> ToolResultBlockParam: """Execute one ``tool_use`` block and emit its runtime facts.""" tool_input = _json_value(block.input) @@ -375,7 +399,10 @@ def _run_one_tool( else: # bash; unknown names have already been rejected command = block.input["command"] with Spinner("Running..."): - output, is_error = run_bash(command) + output, is_error = run_bash(command, **( + {"timeout_seconds": remaining_seconds} + if remaining_seconds is not None else {} + )) except BaseException as exc: emitter.emit( "tool.completed", @@ -466,6 +493,8 @@ def _run_exchange( distinguishing completion, turn-budget exhaustion, and response truncation. """ run_id = f"run-{uuid.uuid4()}" + if time_budget_seconds is not None: + check_deadline_support() run_started_ns = time.perf_counter_ns() projector = _TextOutputProjector(reply_prefix) with EventJournal.create(run_id) as journal: @@ -510,7 +539,10 @@ def _run_exchange( time_budget_seconds=time_budget_seconds, ) except BaseException as exc: - cost_reconciliation = _reconcile_costs(client, journal, emitter) + cost_reconciliation = _reconcile_costs( + client, journal, emitter, + max_seconds=_COST_RECONCILIATION_SECONDS if time_budget_seconds else None, + ) emitter.emit( "run.failed", { @@ -528,7 +560,10 @@ def _run_exchange( ) raise else: - cost_reconciliation = _reconcile_costs(client, journal, emitter) + cost_reconciliation = _reconcile_costs( + client, journal, emitter, + max_seconds=_COST_RECONCILIATION_SECONDS if time_budget_seconds else None, + ) emitter.emit( "run.completed", { @@ -574,20 +609,24 @@ def _run_model_loop( ) while True: model_call_id = f"model-{uuid.uuid4()}" + remaining = None if deadline is not None: remaining = deadline - time.monotonic() if remaining <= 0: return "time_budget_exhausted" + if deadline is not None or (max_turns is not None and retries == 0): _append_budget_note( messages, _time_budget_note( turn=turns + 1, - elapsed=time_budget_seconds - remaining, + elapsed=time_budget_seconds - remaining if deadline is not None else 0, budget=time_budget_seconds, remaining=remaining, + max_turns=max_turns, ), emitter=emitter, model_call_id=model_call_id, + reason="time_budget" if deadline is not None else "turn_budget", ) # A spinner marks the wait for the reply; the first streamed # token replaces it with the reply prefix. A tool-only reply @@ -606,12 +645,13 @@ def _run_model_loop( generation_id = None stream_entered = False try: - with Spinner() as spinner, client.messages.stream( + with wall_clock_limit(remaining), Spinner() as spinner, client.messages.stream( model=model, max_tokens=max_tokens, system=system, tools=TOOLS, messages=messages, + **({"timeout": remaining} if remaining is not None else {}), ) as stream: stream_entered = True generation_id = _response_header(stream, "x-generation-id") @@ -636,17 +676,28 @@ def _run_model_loop( ) message = stream.get_final_message() model_completed_ns = time.perf_counter_ns() + except DeadlineExceeded as exc: + emitter.emit("model.failed", { + "model_call_id": model_call_id, + "error_type": type(exc).__name__, "message": str(exc), + "generation_id": generation_id, + "duration_ms": (time.perf_counter_ns() - model_started_ns) / 1_000_000, + "will_retry": False, "retry_delay_seconds": 0, + "source_timestamp": utc_now(), + }) + return "time_budget_exhausted" except (anthropic.APIError, *HTTP_ERRORS) as exc: now = time.monotonic() if retry_deadline is None: retry_deadline = now + STREAM_RETRY_WINDOW_SECONDS delay = STREAM_RETRY_DELAYS[retries] if retries < len(STREAM_RETRY_DELAYS) else 0 - will_retry = ( + retryable = ( stream_entered and isinstance(exc, RETRYABLE_STREAM_ERRORS) and retries < len(STREAM_RETRY_DELAYS) and now + delay <= retry_deadline ) + will_retry = retryable and (deadline is None or now + delay < deadline) emitter.emit( "model.failed", { @@ -661,9 +712,10 @@ def _run_model_loop( }, ) if not will_retry: + if deadline is not None and (now >= deadline or (retryable and now + delay >= deadline)): + return "time_budget_exhausted" raise - # The window limits when another retry may start. An in-flight - # attempt retains the SDK timeout and any external run deadline. + # The retry delay fits inside both the recovery window and budget. time.sleep(delay) retries += 1 continue @@ -722,6 +774,8 @@ def _run_model_loop( emitter.emit("model.completed", payload) turns += 1 + if deadline is not None and time.monotonic() >= deadline: + return "time_budget_exhausted" if message.stop_reason == "max_tokens": # Do not execute partial tool calls or replay them without results. # Thinking may also be cut off before its signature arrives. Keep @@ -761,12 +815,17 @@ def _run_model_loop( # The reply or a preceding tool consumed the remaining time. # Stop the run without starting another tool or model call. return "time_budget_exhausted" - results.append( - _run_one_tool( - block, emitter, model_call_id, - input_error=input_errors.get(block.id), - ) - ) + remaining = None if deadline is None else deadline - time.monotonic() + try: + with wall_clock_limit(remaining): + result = _run_one_tool( + block, emitter, model_call_id, + input_error=input_errors.get(block.id), + remaining_seconds=remaining, + ) + except DeadlineExceeded: + return "time_budget_exhausted" + results.append(result) messages.append({"role": "user", "content": results}) @@ -774,6 +833,8 @@ def _reconcile_costs( client: anthropic.Anthropic, journal: EventJournal, emitter: EventEmitter, + *, + max_seconds: float | None = None, ) -> list[JsonObject]: """Append resolved OpenRouter costs without affecting the run outcome.""" base_url = getattr(client, "base_url", "") @@ -781,6 +842,7 @@ def _reconcile_costs( if not isinstance(credential, str) or not credential: return [] outcomes: list[JsonObject] = [] + deadline = None if max_seconds is None else time.monotonic() + max_seconds entries = EventJournal.replay(journal.path) already_resolved = { str(entry.payload["generation_id"]) @@ -804,12 +866,17 @@ def _reconcile_costs( ): continue diagnostics: list[JsonObject] = [] - resolved = resolve_generation_cost( - base_url, - generation_id, - credential, - diagnostics=diagnostics, - ) + if deadline is not None and time.monotonic() >= deadline: + break + try: + with wall_clock_limit(None if deadline is None else deadline - time.monotonic()): + resolved = resolve_generation_cost( + base_url, generation_id, credential, diagnostics=diagnostics, + ) + except DeadlineExceeded: + outcomes.append({"generation_id": generation_id, "status": "unresolved", + "attempts": diagnostics, "reason": "finalization_deadline"}) + break if resolved is not None: resolved["source_timestamp"] = utc_now() emitter.emit("model.cost_resolved", resolved) diff --git a/src/nanopycodeagent/bash_tool.py b/src/nanopycodeagent/bash_tool.py index 1909fef..004c849 100644 --- a/src/nanopycodeagent/bash_tool.py +++ b/src/nanopycodeagent/bash_tool.py @@ -4,6 +4,8 @@ result string: stdout, then labelled stderr, then the exit code when non-zero. """ +import os +import signal import subprocess from anthropic.types import ToolParam @@ -36,7 +38,7 @@ } -def run_bash(command: str) -> tuple[str, bool]: +def run_bash(command: str, *, timeout_seconds: float | None = None) -> tuple[str, bool]: """Run ``command`` with ``bash -c`` and return ``(output, is_error)``. ``is_error`` is true only when the tool itself failed — here, a timeout. @@ -47,27 +49,44 @@ def run_bash(command: str) -> tuple[str, bool]: ``/dev/null`` so a command that prompts sees EOF instead of eating the user's keystrokes. - Known trades for simplicity: a background child inherits the output - pipes, so ``some_server &`` blocks until the timeout; the timeout kills - bash itself, not necessarily everything it forked; and text mode - translates ``\\r`` in output to ``\\n`` (universal newlines). + Background children inherit output pipes unless redirected. On POSIX, + timeouts and interruptions kill the command's process group; a normally + completed command leaves background services available to later tools. + Text mode translates ``\\r`` to ``\\n`` (universal newlines). """ - try: - process = subprocess.run( + timeout = BASH_TIMEOUT_SECONDS if timeout_seconds is None else min( + BASH_TIMEOUT_SECONDS, timeout_seconds + ) + if timeout <= 0: + return "[command not started: time budget exhausted]", True + with subprocess.Popen( ["bash", "-c", command], - capture_output=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, text=True, errors="replace", stdin=subprocess.DEVNULL, - timeout=BASH_TIMEOUT_SECONDS, - ) - except subprocess.TimeoutExpired: - return f"[command timed out after {BASH_TIMEOUT_SECONDS} seconds]", True + start_new_session=os.name == "posix", + ) as process: + try: + stdout, stderr = process.communicate(timeout=timeout) + except BaseException as exc: + try: + if os.name == "posix": + os.killpg(process.pid, signal.SIGKILL) + else: + process.kill() + except ProcessLookupError: + pass + process.communicate() + if isinstance(exc, subprocess.TimeoutExpired): + return f"[command timed out after {timeout:g} seconds]", True + raise parts = [] - if stdout := process.stdout.rstrip("\n"): + if stdout := stdout.rstrip("\n"): parts.append(stdout) - if stderr := process.stderr.rstrip("\n"): + if stderr := stderr.rstrip("\n"): parts.append("[stderr]\n" + stderr) if process.returncode != 0: parts.append(f"[exit code: {process.returncode}]") diff --git a/src/nanopycodeagent/cli.py b/src/nanopycodeagent/cli.py index be25d52..8eca0b7 100644 --- a/src/nanopycodeagent/cli.py +++ b/src/nanopycodeagent/cli.py @@ -18,6 +18,7 @@ from .agent import DEFAULT_MAX_TURNS, _package_version, run, run_headless from .settings import DEFAULT_MAX_TOKENS, resolve_max_tokens +from .deadline import check_deadline_support # Reserved by argparse for a misuse of the command line itself, and used here # for the same: a task that cannot be read is a mistake in how the agent was @@ -69,8 +70,9 @@ def _build_parser() -> argparse.ArgumentParser: type=int, metavar="N", help=( - "stop a headless run after this many seconds of wall-clock time, " - "telling the model how much remains; default: no time budget" + "limit headless task work to this many wall-clock seconds " + "(POSIX main thread), with up to 30 additional seconds for cost " + "reconciliation before saving the trajectory; default: no time budget" ), ) parser.add_argument( @@ -152,7 +154,14 @@ def main(argv: list[str] | None = None) -> int: if task is None: if args.trajectory is not None: parser.error("--trajectory requires a headless task") + if args.time_budget_seconds is not None: + parser.error("--time-budget-seconds requires a headless task") return run(max_tokens=max_tokens) + if args.time_budget_seconds is not None: + try: + check_deadline_support() + except ValueError as exc: + parser.error(str(exc)) trajectory_path = _trajectory_path(args.trajectory, parser) return run_headless( task, diff --git a/src/nanopycodeagent/deadline.py b/src/nanopycodeagent/deadline.py new file mode 100644 index 0000000..142ea29 --- /dev/null +++ b/src/nanopycodeagent/deadline.py @@ -0,0 +1,39 @@ +"""Interrupt synchronous work at a wall-clock deadline on POSIX CLI runs.""" + +from contextlib import contextmanager +import signal +import threading + + +class DeadlineExceeded(BaseException): + """A local deadline, which transport and best-effort handlers must not retry.""" + + +def check_deadline_support() -> None: + if not hasattr(signal, "setitimer") or threading.current_thread() is not threading.main_thread(): + raise ValueError("wall-clock budgets require a POSIX main thread") + if signal.getitimer(signal.ITIMER_REAL)[0]: + raise ValueError("wall-clock budgets cannot share an active SIGALRM timer") + + +@contextmanager +def wall_clock_limit(seconds: float | None): + """Bound a blocking operation, restoring the caller's signal handler.""" + if seconds is None: + yield + return + if seconds <= 0: + raise DeadlineExceeded("wall-clock budget exhausted") + check_deadline_support() + previous = signal.getsignal(signal.SIGALRM) + + def expire(signum, frame): + raise DeadlineExceeded("wall-clock budget exhausted") + + signal.signal(signal.SIGALRM, expire) + signal.setitimer(signal.ITIMER_REAL, seconds) + try: + yield + finally: + signal.setitimer(signal.ITIMER_REAL, 0) + signal.signal(signal.SIGALRM, previous) diff --git a/tests/helpers.py b/tests/helpers.py index 9e07a4b..eaab0b8 100644 --- a/tests/helpers.py +++ b/tests/helpers.py @@ -7,6 +7,7 @@ """ import json +from copy import deepcopy from types import SimpleNamespace import anthropic @@ -125,7 +126,7 @@ def __init__(self, script): self.kwargs = [] # full kwargs passed to each stream() call def stream(self, **kwargs): - self.calls.append(list(kwargs["messages"])) # freeze history at call time + self.calls.append(deepcopy(kwargs["messages"])) self.kwargs.append(kwargs) item = self._script.pop(0) if isinstance(item, FakeStream): diff --git a/tests/test_agent_events.py b/tests/test_agent_events.py index 723a592..b83eb58 100644 --- a/tests/test_agent_events.py +++ b/tests/test_agent_events.py @@ -54,6 +54,8 @@ def test_headless_model_reply_is_journaled_without_changing_stdout( assert captured.out == "done\n" entries = EventJournal.replay(_only_journal_path()) + assert [entry.seq for entry in entries] == list(range(1, len(entries) + 1)) + entries = [entry for entry in entries if entry.type != "input.injected"] assert [entry.type for entry in entries] == [ "run.started", "user.message", @@ -62,7 +64,6 @@ def test_headless_model_reply_is_journaled_without_changing_stdout( "model.completed", "run.completed", ] - assert [entry.seq for entry in entries] == [1, 2, 3, 4, 5, 6] assert all("timestamp_source" not in entry.payload for entry in entries) started_run = entries[0].payload @@ -117,6 +118,8 @@ def normalize_after_model_timing(value): assert agent.run_headless("fix it") == 0 entries = EventJournal.replay(_only_journal_path()) + assert [entry.seq for entry in entries] == list(range(1, len(entries) + 1)) + entries = [entry for entry in entries if entry.type != "input.injected"] completed = next(entry for entry in entries if entry.type == "model.completed") assert completed.payload["duration_ms"] == 1 @@ -163,6 +166,8 @@ def resolved(base_url, generation_id, credential, **kwargs): assert agent.run_headless("fix it") == 0 entries = EventJournal.replay(_only_journal_path()) + assert [entry.seq for entry in entries] == list(range(1, len(entries) + 1)) + entries = [entry for entry in entries if entry.type != "input.injected"] assert [entry.type for entry in entries[-3:]] == [ "model.completed", "model.cost_resolved", @@ -251,6 +256,8 @@ def test_failed_tool_events_project_the_existing_tool_output( ) entries = EventJournal.replay(_only_journal_path()) + assert [entry.seq for entry in entries] == list(range(1, len(entries) + 1)) + entries = [entry for entry in entries if entry.type != "input.injected"] assert [entry.type for entry in entries] == [ "run.started", "user.message", @@ -318,6 +325,8 @@ def _chunks(): assert "API error: peer disconnected" in captured.err entries = EventJournal.replay(_only_journal_path()) + assert [entry.seq for entry in entries] == list(range(1, len(entries) + 1)) + entries = [entry for entry in entries if entry.type != "input.injected"] assert [entry.type for entry in entries] == [ "run.started", "user.message", @@ -355,6 +364,8 @@ def raise_from_read(*args, **kwargs): assert captured.out == f"[read] {target}\n" entries = EventJournal.replay(_only_journal_path()) + assert [entry.seq for entry in entries] == list(range(1, len(entries) + 1)) + entries = [entry for entry in entries if entry.type != "input.injected"] assert [entry.type for entry in entries] == [ "run.started", "user.message", diff --git a/tests/test_bash_tool.py b/tests/test_bash_tool.py index 2cf84df..7a7b4c4 100644 --- a/tests/test_bash_tool.py +++ b/tests/test_bash_tool.py @@ -5,6 +5,8 @@ """ from nanopycodeagent import bash_tool +import shlex +import time def test_run_bash_captures_stdout(): @@ -60,3 +62,25 @@ def test_run_bash_truncates_long_output(monkeypatch): assert output == "a" * 10 + "\n[... output truncated ...]" assert is_error is False + + +def test_timeout_kills_children_before_they_can_write(tmp_path): + target = tmp_path / "must-not-exist" + output, is_error = bash_tool.run_bash( + f"(sleep 0.4; touch {shlex.quote(str(target))}) & wait", + timeout_seconds=0.05, + ) + assert is_error and "timed out" in output + time.sleep(0.5) + assert not target.exists() + + +def test_normal_exit_preserves_background_service(tmp_path): + target = tmp_path / "finished" + output, is_error = bash_tool.run_bash( + f"(sleep 0.1; touch {shlex.quote(str(target))}) >/dev/null 2>&1 &", + timeout_seconds=1, + ) + assert not is_error + time.sleep(0.3) + assert target.exists() diff --git a/tests/test_cli.py b/tests/test_cli.py index 3e61177..c1c34e6 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -49,7 +49,8 @@ def test_prompt_argument_runs_the_task_and_exits_zero(monkeypatch, capsys): assert cli.main(["-p", "say hi"]) == 0 - assert messages.calls[0] == [{"role": "user", "content": "say hi"}] + assert messages.calls[0][0]["role"] == "user" + assert messages.calls[0][0]["content"].startswith("say hi\n\n[runtime budget]") # The headless prompt goes out, not the conversational one. assert messages.kwargs[0]["system"] == agent.HEADLESS_SYSTEM_PROMPT out = capsys.readouterr().out @@ -68,7 +69,8 @@ def test_stdin_pipe_is_taken_as_the_task(monkeypatch, capsys): assert cli.main([]) == 0 - assert messages.calls[0] == [{"role": "user", "content": "fix the bug"}] + assert messages.calls[0][0]["role"] == "user" + assert messages.calls[0][0]["content"].startswith("fix the bug\n\n[runtime budget]") def test_prompt_file_is_read_as_the_task(monkeypatch, tmp_path, capsys): @@ -79,7 +81,8 @@ def test_prompt_file_is_read_as_the_task(monkeypatch, tmp_path, capsys): assert cli.main(["--prompt-file", str(task_file)]) == 0 - assert messages.calls[0] == [{"role": "user", "content": "port the parser"}] + assert messages.calls[0][0]["role"] == "user" + assert messages.calls[0][0]["content"].startswith("port the parser\n\n[runtime budget]") def test_banner_stays_off_stdout_in_a_headless_run(monkeypatch, capsys): @@ -108,7 +111,7 @@ def test_trajectory_path_writes_atif_without_changing_headless_stdout( trajectory = json.loads(trajectory_path.read_text(encoding="utf-8")) assert trajectory["schema_version"] == "ATIF-v1.7" assert trajectory["steps"][0]["message"] == "say hi" - assert trajectory["steps"][1]["message"] == "done" + assert next(step for step in trajectory["steps"] if step["source"] == "agent")["message"] == "done" assert trajectory["extra"]["terminal"]["outcome"] == "completed" @@ -144,7 +147,7 @@ def test_partial_model_usage_is_not_reported_as_complete_trajectory_totals( trajectory = json.loads(trajectory_path.read_text(encoding="utf-8")) assert trajectory["final_metrics"] == { - "total_steps": 3, + "total_steps": 5, "extra": { "usage_complete": False, "known_cost_usd": 0.0, @@ -293,8 +296,8 @@ def _gen(): assert captured.out == "partial reply" assert "API error: peer disconnected" in captured.err trajectory = json.loads(trajectory_path.read_text(encoding="utf-8")) - assert trajectory["steps"][1]["message"] == "partial reply" - assert trajectory["steps"][1]["extra"]["incomplete"] is True + assert next(step for step in trajectory["steps"] if step["source"] == "agent")["message"] == "partial reply" + assert next(step for step in trajectory["steps"] if step["source"] == "agent")["extra"]["incomplete"] is True terminal = trajectory["extra"]["terminal"] terminal_summary = { key: terminal[key] diff --git a/tests/test_stream_recovery.py b/tests/test_stream_recovery.py index 3292403..9e3e518 100644 --- a/tests/test_stream_recovery.py +++ b/tests/test_stream_recovery.py @@ -51,8 +51,8 @@ def test_transport_families_recover_and_preserve_failed_attempt( failed = [e for e in entries() if e.type == "model.failed"] assert len(failed) == 1 and failed[0].payload["will_retry"] is True trajectory = json.loads(trajectory_path.read_text()) - assert [s["message"] for s in trajectory["steps"]] == ["task", "partial", "done"] - assert trajectory["steps"][1]["extra"]["incomplete"] is True + assert [s["message"] for s in trajectory["steps"] if not s.get("extra", {}).get("injected")] == ["task", "partial", "done"] + assert next(step for step in trajectory["steps"] if step["source"] == "agent")["extra"]["incomplete"] is True assert trajectory["extra"]["terminal"]["outcome"] == "completed" assert trajectory["final_metrics"]["extra"]["usage_complete"] is False assert trajectory["final_metrics"]["extra"]["cost_is_partial"] is True @@ -181,7 +181,7 @@ def respond(request): failed = next(e for e in entries() if e.type == "model.failed") assert failed.payload["generation_id"] == "gen-2" trajectory = json.loads(trajectory_path.read_text()) - assert [s["extra"].get("incomplete", False) for s in trajectory["steps"]] == [False, False, True, False, False] + assert [s["extra"].get("incomplete", False) for s in trajectory["steps"] if not s.get("extra", {}).get("injected")] == [False, False, True, False, False] def test_interrupted_generation_cost_is_reconciled_without_inventing_usage(monkeypatch, tmp_path): diff --git a/tests/test_time_budget.py b/tests/test_time_budget.py index 5260432..1d61eab 100644 --- a/tests/test_time_budget.py +++ b/tests/test_time_budget.py @@ -7,6 +7,8 @@ """ import json +import signal +import time import pytest @@ -71,7 +73,7 @@ def _texts(messages): def _fake_bash(executions): - def run_bash(command): + def run_bash(command, **kwargs): executions.append(command) return "ok", False @@ -119,12 +121,12 @@ def test_time_budget_is_injected_and_stops_the_run(monkeypatch, capsys): # lets the provider reuse its prefix cache. It rides at the tail instead. systems = [call["system"] for call in messages.kwargs] assert len(set(systems)) == 1 - assert "[time budget]" not in systems[0] + assert "[runtime budget]" not in systems[0] assert "Elapsed 0:00 of 16:40; 16:40 remaining" in _texts(messages.calls[0]) assert "Elapsed 7:30 of 16:40; 9:10 remaining" in _texts(messages.calls[1]) final_text = _texts(messages.calls[2]) assert "Only 1:40 of 16:40 left" in final_text - assert "Stop investigating now" in final_text + assert "Finalize now" in final_text captured = capsys.readouterr() assert "1000s time budget" in captured.err @@ -136,7 +138,7 @@ def test_time_budget_is_injected_and_stops_the_run(monkeypatch, capsys): assert entries[-1].payload["outcome"] == "time_budget_exhausted" -def test_without_a_budget_the_system_prompt_is_unchanged(monkeypatch): +def test_turn_only_budget_is_injected_without_changing_system(monkeypatch): clock = FakeTime() monkeypatch.setattr(agent, "time", clock) monkeypatch.setattr(agent, "run_bash", _fake_bash([])) @@ -145,12 +147,12 @@ def test_without_a_budget_the_system_prompt_is_unchanged(monkeypatch): assert cli.main(["-p", "just answer", "--max-turns", "5"]) == 0 - assert "[time budget]" not in messages.kwargs[0]["system"] - assert "[time budget]" not in _texts(messages.calls[0]) + assert "[runtime budget]" not in messages.kwargs[0]["system"] + assert "Turn 1 of 5" in _texts(messages.calls[0]) entries = _journal_entries() assert entries[0].payload["time_budget_seconds"] is None assert entries[-1].payload["outcome"] == "completed" - assert not any(entry.type == "input.injected" for entry in entries) + assert [entry.payload["reason"] for entry in entries if entry.type == "input.injected"] == ["turn_budget"] @pytest.mark.parametrize("tool_seconds", [1, 2]) @@ -159,7 +161,7 @@ def test_deadline_is_checked_before_each_tool(monkeypatch, tmp_path, tool_second monkeypatch.setattr(agent, "time", clock) executions = [] - def run_bash(command): + def run_bash(command, **kwargs): executions.append(command) clock.advance(tool_seconds) return "first result", False @@ -209,9 +211,9 @@ def test_budget_reminders_survive_journal_and_atif(monkeypatch, tmp_path, struct assert agent._run_exchange( FakeClient(messages), "test-model", [{"role": "user", "content": task}], - agent.HEADLESS_SYSTEM_PROMPT, max_turns=10, time_budget_seconds=1000, + agent.HEADLESS_SYSTEM_PROMPT, max_turns=100, time_budget_seconds=1000, trajectory_path=trajectory_path, - ) == "completed" + ) == "time_budget_exhausted" notes = [] for call in messages.calls: @@ -220,8 +222,8 @@ def test_budget_reminders_survive_journal_and_atif(monkeypatch, tmp_path, struct assert len(notes) == 3 assert "Elapsed 0:00" in notes[0] assert "Elapsed 7:30" in notes[1] - assert "Stop investigating now" in notes[2] - assert "write your best answer to it" in notes[2] + assert "Finalize now" in notes[2] + assert "Complete and save the required deliverables" in notes[2] entries = _journal_entries() user_entry = next(entry for entry in entries if entry.type == "user.message") @@ -269,5 +271,107 @@ def test_finalization_note_escalates_below_the_reserve(): late = agent._time_budget_note( turn=9, elapsed=990, budget=1000, remaining=10 ) - assert "remaining" in early and "Stop investigating now" not in early - assert "Stop investigating now" in late + assert "remaining" in early and "Finalize now" not in early + assert "Finalize now" in late + + +def test_turn_budget_finalizes_while_time_is_plentiful(): + early = agent._time_budget_note(turn=90, max_turns=100, elapsed=1200, budget=3420, remaining=2220) + late = agent._time_budget_note(turn=91, max_turns=100, elapsed=1210, budget=3420, remaining=2210) + assert "Finalize now" not in early + assert "10 replies remaining including this one" in late + assert "Finalize now" in late + assert "its tool calls will not execute" in late + + +def test_stalled_stream_is_closed_and_journaled_at_deadline(monkeypatch, tmp_path): + closed = [] + + class StalledStream(FakeStream): + def __iter__(self): + yield from super().__iter__() + time.sleep(10) + + def __exit__(self, *args): + closed.append(True) + + messages = FakeMessages([StalledStream([text_block("partial")])]) + patch_client(monkeypatch, FakeClient(messages)) + trajectory = tmp_path / "trajectory.json" + started = time.monotonic() + assert agent.run_headless("task", time_budget_seconds=0.1, trajectory_path=trajectory) == 0 + assert time.monotonic() - started < 2 + assert closed == [True] + events = _journal_entries() + failed = next(e for e in events if e.type == "model.failed") + assert failed.payload["error_type"] == "DeadlineExceeded" + assert failed.payload["will_retry"] is False + assert events[-1].payload["outcome"] == "time_budget_exhausted" + assert json.loads(trajectory.read_text())["extra"]["terminal"]["outcome"] == "time_budget_exhausted" + assert signal.getitimer(signal.ITIMER_REAL)[0] == 0 + + +def test_budget_interrupts_a_real_long_command(monkeypatch): + messages = FakeMessages([FakeStream([tool_use_block("slow", "sleep 10")], stop_reason="tool_use")]) + patch_client(monkeypatch, FakeClient(messages)) + started = time.monotonic() + assert agent.run_headless("task", time_budget_seconds=0.1) == 0 + assert time.monotonic() - started < 2 + events = _journal_entries() + completed = next(e for e in events if e.type == "tool.completed") + assert completed.payload["is_error"] is True + assert events[-1].payload["outcome"] == "time_budget_exhausted" + + +def test_retry_delay_cannot_spend_the_remaining_budget(monkeypatch): + import httpx + from test_stream_recovery import BrokenStream + + sleeps = [] + monkeypatch.setattr(agent.time, "sleep", sleeps.append) + messages = FakeMessages([BrokenStream(httpx.ReadError("interrupted"))]) + patch_client(monkeypatch, FakeClient(messages)) + assert agent.run_headless("task", time_budget_seconds=0.5) == 0 + assert sleeps == [] + assert len(messages.calls) == 1 + assert _journal_entries()[-1].payload["outcome"] == "time_budget_exhausted" + + +def test_cost_reconciliation_has_one_shared_finalization_budget(monkeypatch, tmp_path): + calls = [] + + def stalled_lookup(*args, **kwargs): + calls.append(args[1]) + time.sleep(10) + + monkeypatch.setattr(agent, "resolve_generation_cost", stalled_lookup) + monkeypatch.setattr(agent, "_COST_RECONCILIATION_SECONDS", 0.1) + messages = FakeMessages([ + FakeStream([tool_use_block("one", "true")], stop_reason="tool_use", response_headers={"x-generation-id": "gen-one"}), + FakeStream([text_block("done")], response_headers={"x-generation-id": "gen-two"}), + ]) + patch_client(monkeypatch, FakeClient(messages)) + started = time.monotonic() + trajectory = tmp_path / "trajectory.json" + assert agent.run_headless("task", time_budget_seconds=10, trajectory_path=trajectory) == 0 + assert time.monotonic() - started < 2 + assert calls == ["gen-one"] + assert _journal_entries()[-1].payload["outcome"] == "completed" + assert json.loads(trajectory.read_text())["final_metrics"]["extra"]["cost_is_partial"] is True + + +def test_unbudgeted_interactive_exchange_has_no_reminders(monkeypatch): + messages = FakeMessages([[text_block("done")]]) + assert agent._run_exchange(FakeClient(messages), "test", [{"role": "user", "content": "hi"}], agent.SYSTEM_PROMPT) == "completed" + assert not any(e.type == "input.injected" for e in _journal_entries()) + + +def test_deadline_restores_handler_after_interruption(): + from nanopycodeagent.deadline import DeadlineExceeded, wall_clock_limit + + handler = signal.getsignal(signal.SIGALRM) + with pytest.raises(DeadlineExceeded): + with wall_clock_limit(0.02): + time.sleep(10) + assert signal.getsignal(signal.SIGALRM) == handler + assert signal.getitimer(signal.ITIMER_REAL) == (0, 0) diff --git a/tests/test_tool_validation.py b/tests/test_tool_validation.py index 49b855a..f13ba86 100644 --- a/tests/test_tool_validation.py +++ b/tests/test_tool_validation.py @@ -69,7 +69,7 @@ def must_not_run(*args, **kwargs): assert agent.run_headless("work", trajectory_path=trajectory_path) == 0 - result, = messages.calls[1][-1]["content"] + result, = [block for block in messages.calls[1][-1]["content"] if block["type"] == "tool_result"] assert result["tool_use_id"] == "bad" and result["is_error"] is True assert diagnostic in result["content"] assert "not executed" in capsys.readouterr().out @@ -79,7 +79,7 @@ def must_not_run(*args, **kwargs): assert completed.payload["error"]["type"] == "ToolInputError" assert entries[-1].payload["outcome"] == "completed" trajectory = json.loads(trajectory_path.read_text()) - step = trajectory["steps"][1] + step = next(step for step in trajectory["steps"] if step["source"] == "agent") observation, = step["observation"]["results"] assert observation["source_call_id"] == "bad" assert observation["extra"]["is_error"] is True @@ -113,7 +113,7 @@ def test_model_corrects_bad_call_and_keeps_successful_sibling( assert agent.run_headless("work", max_turns=3) == 0 assert target.read_text() == "corrected" - results = messages.calls[1][-1]["content"] + results = [block for block in messages.calls[1][-1]["content"] if block["type"] == "tool_result"] assert [r["tool_use_id"] for r in results] == ["good", "bad"] assert [r["is_error"] for r in results] == [False, True] assert messages.calls[2][-1]["content"][0]["is_error"] is False @@ -220,7 +220,7 @@ def respond(request): assert requests[1]["messages"][-2]["content"][0]["input"] == {} assert requests[2]["messages"][-1]["content"][0]["is_error"] is False trajectory = json.loads(trajectory_path.read_text()) - call = trajectory["steps"][1]["tool_calls"][0] + call = next(step for step in trajectory["steps"] if step["source"] == "agent")["tool_calls"][0] if bad_json.startswith("{") and bad_json != "{}": assert call["extra"]["input_json"] == bad_json assert "JSON" in rejected["content"] diff --git a/tests/test_truncation.py b/tests/test_truncation.py index 96451f8..32222a0 100644 --- a/tests/test_truncation.py +++ b/tests/test_truncation.py @@ -87,7 +87,7 @@ def resolve(base_url, generation_id, credential, **kwargs): trajectory = json.loads(trajectory_path.read_text()) assert trajectory["schema_version"] == "ATIF-v1.7" assert trajectory["extra"]["terminal"]["outcome"] == "response_truncated" - assert trajectory["steps"][1]["extra"]["stop_reason"] == "max_tokens" + assert next(step for step in trajectory["steps"] if step["source"] == "agent")["extra"]["stop_reason"] == "max_tokens" assert trajectory["final_metrics"]["total_prompt_tokens"] == 10 assert trajectory["agent"]["extra"]["max_tokens"] == max_tokens assert trajectory["final_metrics"]["total_completion_tokens"] == max_tokens @@ -178,8 +178,8 @@ def respond(request): assert completed.payload["tool_calls"][0]["input"] == {} assert not any(entry.type.startswith("tool.") for entry in entries) trajectory = json.loads((tmp_path / "trajectory.json").read_text()) - assert trajectory["steps"][1]["tool_calls"][0]["arguments"] == {} - assert "observation" not in trajectory["steps"][1] + assert next(step for step in trajectory["steps"] if step["source"] == "agent")["tool_calls"][0]["arguments"] == {} + assert "observation" not in next(step for step in trajectory["steps"] if step["source"] == "agent") @pytest.mark.parametrize("schema_version", [1, 2, 3, 4]) From 6d7042d367ef7275293bead0164b76e43ebd0e53 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Tue, 29 Sep 2026 21:32:19 +0800 Subject: [PATCH 09/12] docs: describe dual-budget finalization and deadline behavior --- benchmarks/harbor/README.md | 7 +++ benchmarks/harbor/README.zh-CN.md | 5 +++ docs/changelogs/0.8.x.md | 14 ++++-- docs/dev_docs/en/event-journal-protocol-v4.md | 43 ++++++++++++------- .../zh-CN/event-journal-protocol-v4.md | 19 +++++--- docs/user_docs/en/cli_reference.md | 23 ++++++++-- docs/user_docs/zh-CN/cli_reference.md | 20 +++++++-- 7 files changed, 99 insertions(+), 32 deletions(-) diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index de342bb..fdd41cd 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -45,6 +45,13 @@ container's current directory, and saves combined stdout/stderr to `/logs/agent/nanopycodeagent.txt`. It uses the CLI's 50-turn default; override that with `--agent-kwarg max_turns=20`. +Use `--agent-kwarg time_budget_seconds=N` to bound task work and remind the model +to finalize before either time or replies run out. Choose a budget inside each +task's native timeout, reserving at least 30 seconds for cost reconciliation +plus trajectory writing and harness overhead. For example, the 3600-second +targeted experiments use 3420 seconds; that value does not fit a 900-second task. +No time budget is inferred automatically by the adapter. + For the per-reply generation limit, pass `--agent-kwarg max_tokens=32768`. This becomes `--max-tokens 32768` in the container and overrides the forwarded `ANTHROPIC_MAX_TOKENS` environment variable. When omitted, the adapter sends no diff --git a/benchmarks/harbor/README.zh-CN.md b/benchmarks/harbor/README.zh-CN.md index 15620c4..3b84682 100644 --- a/benchmarks/harbor/README.zh-CN.md +++ b/benchmarks/harbor/README.zh-CN.md @@ -41,6 +41,11 @@ adapter 通过 stdin 发送任务指令,在 task 容器的当前目录中运 的 stdout/stderr 保存到 `/logs/agent/nanopycodeagent.txt`。它默认沿用 CLI 的 50 轮限制;可以通过 `--agent-kwarg max_turns=20` 覆盖此设置。 +可用 `--agent-kwarg time_budget_seconds=N` 限制解题时间,并在时间或轮数接近耗尽时 +提醒模型收尾。预算应落在每题原生时限内,至少预留 30 秒费用补查时间,以及轨迹写入 +和 harness 开销。例如,本轮原生 3600 秒的针对性实验使用 3420 秒;该值不能用于 +900 秒题目。adapter 不会自动推导时间预算。 + 每次回复的生成上限可通过 `--agent-kwarg max_tokens=32768` 指定,它会转换为容器内的 `--max-tokens 32768`,优先于透传的 `ANTHROPIC_MAX_TOKENS` 环境变量。未指定时, adapter 不添加该 flag,沿用已安装 agent 的环境变量/配置文件/默认值(引入该参数的 diff --git a/docs/changelogs/0.8.x.md b/docs/changelogs/0.8.x.md index 9b4805e..0d07724 100644 --- a/docs/changelogs/0.8.x.md +++ b/docs/changelogs/0.8.x.md @@ -11,10 +11,13 @@ All notable changes in the **0.8.x** release series are documented here. trajectories while retaining compatibility with older journals. - Add a wall-clock time budget to headless runs through `--time-budget-seconds` and the Harbor adapter `time_budget_seconds` kwarg. - Tell the model how much time remains each turn, escalate to a stop-and-write - warning inside a reserved finalization window, recheck the deadline before - every tool, and stop with a `time_budget_exhausted` outcome before the - harness timeout can kill the process. Append the reminder to the tail of the + Tell the model how much time and how many replies remain, and start + finalization guidance when either reserve is reached (10 replies or the + larger of 180 seconds and 15% of the time budget). On POSIX main threads, + interrupt in-flight model/tool work at the deadline, constrain retry waits, + and limit subsequent cost reconciliation to 30 seconds before saving the + trajectory. Budget exhaustion has a distinct `time_budget_exhausted` outcome. + Append the reminder to the tail of the conversation so the system prompt stays fixed and provider prefix caching keeps working, and record each injected reminder as an `input.injected` event projected into ATIF. Record the budget in startup output, Event @@ -22,6 +25,9 @@ All notable changes in the **0.8.x** release series are documented here. supported. ### Fixed +- Terminate a timed-out or interrupted bash command's process group on POSIX, + preventing its children from continuing after the deadline. Preserve + background services when their launching command finishes normally. - Return correctable tool errors for missing or incorrectly typed arguments and unknown tool names, allowing the model to retry within the existing turn budget. Keep previews safe, reject incomplete argument JSON yielded by the diff --git a/docs/dev_docs/en/event-journal-protocol-v4.md b/docs/dev_docs/en/event-journal-protocol-v4.md index 31c89db..f4c37f4 100644 --- a/docs/dev_docs/en/event-journal-protocol-v4.md +++ b/docs/dev_docs/en/event-journal-protocol-v4.md @@ -5,24 +5,28 @@ > Do not edit by hand. The current writer emits `schema_version = 4` for all runs. Readers and the -ATIF projector continue to accept v1, v2, and v3. Public trajectories remain +ATIF projector continue to accept v1, v2, and v3; public trajectories remain ATIF-v1.7. All contracts from [v3](event-journal-protocol-v3.md) still apply except for the changes below. ## Time budget outcome `run.completed.payload.outcome` adds `time_budget_exhausted`: the core observed -an exhausted wall-clock budget before starting another model or tool call. -The meanings of `completed`, `max_turns_exhausted`, and `response_truncated` -remain unchanged. +an exhausted wall-clock budget before another model or tool call, after a +model response, or during execution. The meanings of `completed`, +`max_turns_exhausted`, and `response_truncated` remain unchanged. A headless run with a time budget computes its deadline using a monotonic clock. It checks before each model attempt and before executing every tool in a reply. After one tool consumes the remaining budget, subsequent tools -do not start. An in-flight model or tool call is not interrupted and may -outlast the deadline. Existing precedence for response truncation and the -turn budget is unchanged; a complete final reply needing no further work -can still end normally. +do not start. A real-time signal timer on a POSIX main thread interrupts +in-flight model or tool work. Configuring a time budget is explicitly rejected +when the timer is unsupported, execution is outside the main thread, or an +existing real-time timer is active. A model reply returned after the deadline +records time-budget exhaustion even if it requests no further tools. An +interrupted model stream records a non-retried `model.failed` with error type +`DeadlineExceeded`; an interrupted tool records `tool.completed` with an error. +Partial output is not presented as a complete model response. The optional `run.started.time_budget_seconds` field records the configured positive integer number of seconds, or null when no budget is configured. @@ -30,10 +34,12 @@ Older journals may omit this field. ATIF preserves it in `agent.extra.time_budget_seconds`. Budget exhaustion finalizes normally with `run.completed` and headless exit -code `0`; cost reconciliation and trajectory writing still run. ATIF records -`extra.terminal.status = "completed"` and -`extra.terminal.outcome = "time_budget_exhausted"`. Consumers must inspect -the outcome rather than infer task completion from status alone. +code `0`; cost reconciliation and trajectory writing still run. For a +time-budgeted run, cost reconciliation has a separate shared 30-second limit. +Unresolved costs remain pending, and the trajectory is then written. ATIF +records `extra.terminal.status = "completed"` and +`extra.terminal.outcome = "time_budget_exhausted"`. Consumers must inspect the +outcome rather than infer task completion from status alone. `model.completed` retains all tools requested by the reply. Only tools that actually execute have `tool.started`, `tool.completed`, and corresponding @@ -50,16 +56,21 @@ payload fields are: | --- | --- | --- | | `model_call_id` | nonempty string | The upcoming model attempt that will use this input, matching the immediately following `model.started`. | | `content` | string | The complete text appended this time, not a copy of the whole conversation or tool results. | -| `reason` | nonempty string | Reason for injection; time budget reminders use `time_budget`. | +| `reason` | nonempty string | Reason for injection: `time_budget` when a time budget is configured, or `turn_budget` when only the reply count is limited. | | `source_timestamp` | RFC 3339 UTC or null | Time of injection. | With a time budget configured, every model attempt, including retries, appends a reminder to the most recent user message and records this event before `model.started`. The first reminder follows the task text, subsequent reminders follow tool results, and retries append to the same message. The system prompt -stays unchanged. Each addition is recorded separately, including the final -warning to stop investigating and write the output file. Original user input -and tool results are not rewritten in the Journal. +stays unchanged. Each addition is recorded separately. With only a turn budget, +a reminder is appended before each new reply; transport retries reuse that +reminder. When 10 replies remain (including the upcoming reply), or remaining +time is no greater than `max(180 seconds, total budget * 15%)`, the reminder +instructs the model to complete and save required deliverables, perform +necessary checks, and summarize and stop. It also states that tools requested +by the final allowed reply will not execute. Original user input and tool +results are not rewritten in the Journal. ATIF creates a `source = "user"` step for each `input.injected` in Journal order, with the reminder text in `message`. Here, user denotes the role in the diff --git a/docs/dev_docs/zh-CN/event-journal-protocol-v4.md b/docs/dev_docs/zh-CN/event-journal-protocol-v4.md index ad8c550..30a07f4 100644 --- a/docs/dev_docs/zh-CN/event-journal-protocol-v4.md +++ b/docs/dev_docs/zh-CN/event-journal-protocol-v4.md @@ -10,20 +10,24 @@ ## 时间预算终态 `run.completed.payload.outcome` 新增 `time_budget_exhausted`,表示 core 在启动 -下一次模型调用或工具调用前发现 wall-clock 预算已耗尽。原有 `completed`、 +下一次模型调用或工具调用前、模型返回后或执行期间发现 wall-clock 预算已耗尽。原有 `completed`、 `max_turns_exhausted`、`response_truncated` 的含义保持不变。 配置了时间预算的 headless run 使用单调时钟计算截止时间。检查发生在每次模型 尝试之前,以及同一回复中每一个工具执行之前。一个工具耗尽剩余预算后,后续工具 -不会启动。正在进行的模型或工具调用不会因此被中止,仍可能超过截止时间。 -回复截断和轮数预算的既有优先级保持不变;无需继续工作的完整最终回复仍可正常结束。 +不会启动。POSIX 主线程中的实时信号定时器会中断正在进行的模型或工具工作;不支持 +该定时器、非主线程或已有实时定时器时,显式拒绝配置时间预算。模型回复在期限之后 +返回时,即使不再请求工具,也记录时间预算耗尽。模型流被中断时记录不重试的 +`model.failed`,错误类型为 `DeadlineExceeded`;被中断工具以带错误的 +`tool.completed` 记录。已输出的部分内容不会伪装成完整模型回复。 `run.started` 的可选字段 `time_budget_seconds` 记录配置的正整数秒数;未配置时 为 null。旧 Journal 可以不包含该字段。ATIF 将其保留在 `agent.extra.time_budget_seconds`。 预算耗尽以 `run.completed` 正常收尾,headless 退出码为 `0`,费用补查和轨迹写入 -仍会执行。ATIF 的 `extra.terminal.status` 为 `completed`, +仍会执行;时间预算 run 的费用补查另有共享的 30 秒上限,未补齐费用保持 pending, +随后写出轨迹。ATIF 的 `extra.terminal.status` 为 `completed`, `extra.terminal.outcome` 为 `time_budget_exhausted`。消费者必须检查 outcome, 不能仅凭 status 推断任务完成。 @@ -40,13 +44,16 @@ | --- | --- | --- | | `model_call_id` | 非空字符串 | 即将使用该输入的模型尝试标识,与紧随其后的 `model.started` 对应。 | | `content` | 字符串 | 本次追加的完整文本,不是整个对话或工具结果的副本。 | -| `reason` | 非空字符串 | 注入原因;时间预算提醒使用 `time_budget`。 | +| `reason` | 非空字符串 | 注入原因;配置时间预算时使用 `time_budget`,只有轮数预算时使用 `turn_budget`。 | | `source_timestamp` | RFC 3339 UTC 或 null | 注入发生时间。 | 配置时间预算时,每次模型尝试(包括重试)都会将提醒追加到最近的用户消息尾部, 然后在 `model.started` 之前记录该事件。首次提醒追加到任务文本,后续提醒追加到 工具结果之后;重试时追加到同一个消息尾部。系统提示保持不变。每次追加都单独 -记录,包括要求停止调查并写入输出文件的最终阶段警告。原始用户输入和工具结果 +记录。只有轮数预算时,每次新的回复前追加提醒,传输重试沿用原提醒。剩余 10 次 +回复(含即将开始的回复),或剩余时间不大于 `max(180 秒, 总预算 × 15%)` 时, +提醒要求完成并保存必要产物、执行必要验证并总结结束;同时说明最后一次回复中的 +工具不会执行。原始用户输入和工具结果 不会在 Journal 中被改写。 ATIF 按 Journal 顺序为每个 `input.injected` 创建一个 `source = "user"` 的 diff --git a/docs/user_docs/en/cli_reference.md b/docs/user_docs/en/cli_reference.md index a45318d..54db042 100644 --- a/docs/user_docs/en/cli_reference.md +++ b/docs/user_docs/en/cli_reference.md @@ -11,7 +11,7 @@ task. ```text nanoPyCodeAgent [-h] [-p TEXT | --prompt-file PATH] [--max-turns N] [--max-tokens N] - [--trajectory PATH] [--version] + [--time-budget-seconds N] [--trajectory PATH] [--version] ``` ## Modes and task input @@ -27,7 +27,7 @@ nanoPyCodeAgent Enter `/exit`, press Ctrl-D, or press Ctrl-C at the `You>` prompt to end the session normally. `--max-turns` does not limit interactive exchanges. -`--trajectory` is not available in interactive mode. +`--trajectory` and `--time-budget-seconds` are not available in interactive mode. ### Headless mode @@ -72,6 +72,7 @@ working directory. | `--prompt-file PATH` | — | Read one headless task from a UTF-8 file. The file must be readable and contain a non-empty task. | | `--max-turns N` | `50` | Allow at most `N` model replies in a headless run. `N` must be an integer of at least `1`. | | `--max-tokens N` | `ANTHROPIC_MAX_TOKENS` or `32768` | Maximum generated tokens per model reply in either mode. `N` must be a positive integer; the CLI value overrides environment and settings-file values. | +| `--time-budget-seconds N` | disabled | Limit headless task work to `N` wall-clock seconds. `N` must be a positive integer; requires a POSIX main thread without an active real-time alarm. | | `--trajectory PATH` | disabled | Write the headless run as one ATIF-v1.7 JSON document. See [Trajectory output](#trajectory-output). | | `--version` | — | Print `nanoPyCodeAgent VERSION` and exit successfully. | @@ -80,6 +81,20 @@ calls. If reply `N` still requests tools, those tools are not run because no reply remains to consume their results. Reaching the limit prints a diagnostic to stderr but is still a normal headless exit. +Before each new reply, the model receives a reminder of the remaining replies. +At 10 remaining replies (including the upcoming reply), it is instructed to +save required deliverables, perform necessary checks, and finish with a summary. +The system prompt remains fixed; reminders are appended to the conversation. + +With `--time-budget-seconds`, reminders also report remaining time. Finalization +guidance starts when either the reply reserve or the time reserve is reached; +the time reserve is the larger of 180 seconds and 15% of the configured budget. +The deadline interrupts in-flight model/tool work, and retries must fit in the +remaining time. An interrupted bash command's process group is terminated. +The outcome is `time_budget_exhausted`, with exit status `0`. Cost reconciliation +then has a separate shared 30-second limit before trajectory writing; unresolved +costs remain pending. Leave room for this finalization inside an external timeout. + Each reply has a separate generation limit, defaulting to **32768** tokens. Use `--max-tokens 65536` to override it for one invocation; use `ANTHROPIC_MAX_TOKENS` in the environment or settings file for a persistent @@ -108,7 +123,9 @@ that reply. This window does not cancel an in-flight request: the SDK timeout and any external run deadline still apply. Failures before a response stream opens remain subject to the SDK's own retry policy. Authentication errors, invalid requests, programming errors, and user interrupts are not retried by -this recovery loop. Exhausted transport failures exit headless mode with `1`. +this recovery loop. A configured wall-clock budget also bounds in-flight +requests and retry waits. Exhausted transport failures exit headless mode with +`1`; stopping for the wall-clock budget exits `0` with its distinct outcome. ## Output channels diff --git a/docs/user_docs/zh-CN/cli_reference.md b/docs/user_docs/zh-CN/cli_reference.md index 062758c..1c4581a 100644 --- a/docs/user_docs/zh-CN/cli_reference.md +++ b/docs/user_docs/zh-CN/cli_reference.md @@ -10,7 +10,7 @@ ```text nanoPyCodeAgent [-h] [-p TEXT | --prompt-file PATH] [--max-turns N] [--max-tokens N] - [--trajectory PATH] [--version] + [--time-budget-seconds N] [--trajectory PATH] [--version] ``` ## 模式与任务输入 @@ -24,7 +24,8 @@ nanoPyCodeAgent ``` 输入 `/exit`,或在 `You>` 提示符下按 Ctrl-D 或 Ctrl-C,都可以正常结束会话。 -`--max-turns` 不限制交互会话中的 exchange。交互模式不能使用 `--trajectory`。 +`--max-turns` 不限制交互会话中的 exchange。交互模式不能使用 `--trajectory` +或 `--time-budget-seconds`。 ### Headless 模式 @@ -64,6 +65,7 @@ nanoPyCodeAgent -p "fix the failing tests" | `--prompt-file PATH` | 无 | 从 UTF-8 文件读取一次 headless 任务;文件必须可读并包含非空任务。 | | `--max-turns N` | `50` | 一次 headless run 最多允许 `N` 轮模型回复;`N` 必须是大于或等于 `1` 的整数。 | | `--max-tokens N` | `ANTHROPIC_MAX_TOKENS` 或 `32768` | 两种模式下每次模型回复的最大生成 token 数。`N` 必须是正整数;CLI 值优先于环境变量与 settings 文件。 | +| `--time-budget-seconds N` | 禁用 | 将 headless 解题工作限制在 `N` 秒墙钟时间内。`N` 必须是正整数;要求 POSIX 主线程且没有已启用的实时信号定时器。 | | `--trajectory PATH` | 禁用 | 把 headless run 写成一份 ATIF-v1.7 JSON 文档;参见[Trajectory 输出](#trajectory-输出)。 | | `--version` | 无 | 打印 `nanoPyCodeAgent VERSION` 并成功退出。 | @@ -71,6 +73,17 @@ nanoPyCodeAgent -p "fix the failing tests" 工具,这些工具不会执行,因为已经没有下一轮回复可以使用工具结果。达到上限时,命令 会在 stderr 打印诊断,但仍属于一次正常的 headless 退出。 +每次新回复前,模型都会收到剩余回复次数提醒。剩余 10 次回复时(包含即将开始的 +这次),提醒要求保存必要产物、执行必要验证并以总结结束。系统提示保持固定, +动态提醒追加到对话尾部。 + +指定 `--time-budget-seconds` 后,提醒也会报告剩余时间。轮数或时间任一进入保留 +窗口就提示收尾;时间窗口取 180 秒与总预算 15% 中的较大值。截止时间会中断正在 +进行的模型/工具工作,重试等待也必须落在剩余时间内。被中断的 bash 命令会终止 +整个进程组。终态为 `time_budget_exhausted`,退出码仍为 `0`。随后费用补查有独立、 +共享的 30 秒上限,再写出轨迹;无法补齐的费用保持 pending。外部时限需要为这些 +收尾工作留出空间。 + 每次回复还受独立的生成上限约束,默认为 **32768** tokens。可用 `--max-tokens 65536` 覆盖本次调用,或通过环境变量、settings 文件中的 `ANTHROPIC_MAX_TOKENS` 设置持久默认值。所选模型与 provider 必须接受该上限。 @@ -93,7 +106,8 @@ nanoPyCodeAgent -p "fix the failing tests" 从该轮首次中断起超过 300 秒后,不再安排新的重试。这个窗口不会取消进行中的 请求:SDK 超时和外部执行期限仍然有效。响应流打开前的失败由 SDK 自身的重试 策略处理。这个恢复循环不会重试认证失败、无效请求、程序错误或用户中断。 -传输重试耗尽时,headless 模式退出码为 `1`。 +配置墙钟预算后,正在进行的请求和重试等待也受预算约束。传输重试耗尽时,headless +模式退出码为 `1`;因墙钟预算停止时退出 `0`,并记录独立的预算耗尽终态。 ## 输出通道 From 5df0ca1c5120dfafd7c0b64fe482eeaedd63dbf8 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Tue, 29 Sep 2026 21:36:39 +0800 Subject: [PATCH 10/12] docs(benchmarks): specify the pilot20 acceptance gate --- benchmarks/harbor/reports/dual-budget-20260929.md | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/benchmarks/harbor/reports/dual-budget-20260929.md b/benchmarks/harbor/reports/dual-budget-20260929.md index 2f8a903..9fa577d 100644 --- a/benchmarks/harbor/reports/dual-budget-20260929.md +++ b/benchmarks/harbor/reports/dual-budget-20260929.md @@ -50,10 +50,14 @@ and endpoint-specific restrictions in its The treatment must score reward 1 and end as native `completed` on all six targeted trials, with no supervisor deadline, forced kill, or reconstructed trajectory. Provider auditing must succeed. Compare per-task cost and duration -against the repeated control before proceeding; investigate a material -regression rather than selecting the best run. If the gate passes, run the +against the repeated control before proceeding: a per-task mean cost or +duration above 1.5 times its control mean holds the pipeline for investigation. +Incomplete cost/provider receipts also hold the pipeline. Do not select the +best run. If the gate passes, run the original pinned pilot20 selection once with the same model and provider, 100 -replies, and task-specific budgets inside each native timeout. Historical +replies, and task-specific budgets of native timeout minus 180 seconds. Its +supervisor interrupts at native timeout minus 120 seconds and retains 120 +seconds of grace. Historical pilot20 used another model/routing configuration and is not a controlled estimate of this implementation's effect. From 380271dfd5e84b4dcc9c7059e71e7ca154c31554 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Tue, 29 Sep 2026 22:49:26 +0800 Subject: [PATCH 11/12] fix(bash): bound cleanup after command timeouts --- .../harbor/reports/dual-budget-20260929.md | 31 ++++++-- docs/changelogs/0.8.x.md | 8 ++- docs/user_docs/en/cli_reference.md | 5 +- docs/user_docs/zh-CN/cli_reference.md | 4 +- src/nanopycodeagent/bash_tool.py | 70 ++++++++++++++----- tests/test_bash_tool.py | 34 +++++++++ 6 files changed, 124 insertions(+), 28 deletions(-) diff --git a/benchmarks/harbor/reports/dual-budget-20260929.md b/benchmarks/harbor/reports/dual-budget-20260929.md index 9fa577d..592e24a 100644 --- a/benchmarks/harbor/reports/dual-budget-20260929.md +++ b/benchmarks/harbor/reports/dual-budget-20260929.md @@ -12,8 +12,10 @@ The ordinary CLI and Harbor defaults remain 50 replies and 32768 output tokens. Configured wall-clock budgets interrupt blocking model streams and tool work using a POSIX main-thread timer. Socket and bash timeouts also use the remaining -budget. Bash interruptions terminate the command's process group; normal exits -preserve background services. Stream retries must fit inside the remaining +budget. Bash interruptions terminate the command's process group and, on Linux, +other members of its session. Cleanup closes output pipes instead of waiting +for detached descendants to close them; reaping the direct child is limited to +one second. Normal exits preserve background services. Stream retries must fit inside the remaining budget. Cost reconciliation has a separate shared 30-second cap, after which unresolved costs remain pending and the native Journal/ATIF can be finalized. Callers with an existing real-time alarm, non-main threads, and platforms @@ -65,10 +67,29 @@ These small repeated samples can reveal regressions but do not establish a general optimal default. Provider load, cache state, and sampling remain sources of variation despite a fixed endpoint. -## Local validation before live trials +## Revision during live trials -The pinned benchmark environment passes 351 tests across the main package and +The original treatment (`847cc9895`) exposed a subprocess cleanup defect during +its second Caffe trial. The model invoked `timeout 600 wget ...`; GNU `timeout` +created another process group. Killing the shell's group at 120 seconds left +that child holding the output pipes, so the subsequent unbounded `communicate()` +waited for it. This observation blocks pilot20 independently of reward. + +The repair kills other members of the command's session on Linux, closes the +pipe readers without draining them, and bounds direct-child reaping. The first +two treatment repetitions remain preserved as superseded evidence; the third +was cancelled before launch. The three originally planned control repetitions +remain the comparison baseline. All three treatment repetitions will run again +with the repaired, committed wheel and the same provider/configuration. They +will be evaluated together against the same registered gate. This revision +changes the originally alternating execution order; it does not select the +best treatment trials or discard the defect from the report. + +## Local validation + +The pinned benchmark environment passes 354 tests across the main package and Harbor adapter. Tests cover early turn-based finalization, stable system prompts, reminder journaling, stalled streams, real long-running commands, -child-process termination, preservation of background services after normal +child-process termination (including GNU `timeout` creating another process +group), bounded cleanup with detached pipe holders, preservation of background services after normal exit, bounded retries, bounded cost reconciliation, and ATIF compatibility. diff --git a/docs/changelogs/0.8.x.md b/docs/changelogs/0.8.x.md index 0d07724..8107711 100644 --- a/docs/changelogs/0.8.x.md +++ b/docs/changelogs/0.8.x.md @@ -25,9 +25,11 @@ All notable changes in the **0.8.x** release series are documented here. supported. ### Fixed -- Terminate a timed-out or interrupted bash command's process group on POSIX, - preventing its children from continuing after the deadline. Preserve - background services when their launching command finishes normally. +- Terminate a timed-out or interrupted bash command's process group on POSIX + and other members of its session on Linux, including GNU `timeout` children + in separate groups. Close output pipes without draining detached children + and bound direct-child reaping. Preserve background services when their + launching command finishes normally. - Return correctable tool errors for missing or incorrectly typed arguments and unknown tool names, allowing the model to retry within the existing turn budget. Keep previews safe, reject incomplete argument JSON yielded by the diff --git a/docs/user_docs/en/cli_reference.md b/docs/user_docs/en/cli_reference.md index 54db042..4dd0710 100644 --- a/docs/user_docs/en/cli_reference.md +++ b/docs/user_docs/en/cli_reference.md @@ -90,7 +90,10 @@ With `--time-budget-seconds`, reminders also report remaining time. Finalization guidance starts when either the reply reserve or the time reserve is reached; the time reserve is the larger of 180 seconds and 15% of the configured budget. The deadline interrupts in-flight model/tool work, and retries must fit in the -remaining time. An interrupted bash command's process group is terminated. +remaining time. An interrupted bash command's process group is terminated; +Linux also terminates other members of the command's session. Cleanup closes +output pipes without waiting for detached children, with up to one second to +reap the direct child. The outcome is `time_budget_exhausted`, with exit status `0`. Cost reconciliation then has a separate shared 30-second limit before trajectory writing; unresolved costs remain pending. Leave room for this finalization inside an external timeout. diff --git a/docs/user_docs/zh-CN/cli_reference.md b/docs/user_docs/zh-CN/cli_reference.md index 1c4581a..6fc1907 100644 --- a/docs/user_docs/zh-CN/cli_reference.md +++ b/docs/user_docs/zh-CN/cli_reference.md @@ -80,7 +80,9 @@ nanoPyCodeAgent -p "fix the failing tests" 指定 `--time-budget-seconds` 后,提醒也会报告剩余时间。轮数或时间任一进入保留 窗口就提示收尾;时间窗口取 180 秒与总预算 15% 中的较大值。截止时间会中断正在 进行的模型/工具工作,重试等待也必须落在剩余时间内。被中断的 bash 命令会终止 -整个进程组。终态为 `time_budget_exhausted`,退出码仍为 `0`。随后费用补查有独立、 +整个进程组;Linux 下也会终止同一会话内的其他进程。清理时直接关闭输出管道, +不会等待脱离会话的子进程关闭管道,并最多等待一秒回收直接子进程。 +终态为 `time_budget_exhausted`,退出码仍为 `0`。随后费用补查有独立、 共享的 30 秒上限,再写出轨迹;无法补齐的费用保持 pending。外部时限需要为这些 收尾工作留出空间。 diff --git a/src/nanopycodeagent/bash_tool.py b/src/nanopycodeagent/bash_tool.py index 004c849..49f6b89 100644 --- a/src/nanopycodeagent/bash_tool.py +++ b/src/nanopycodeagent/bash_tool.py @@ -5,6 +5,7 @@ """ import os +from pathlib import Path import signal import subprocess @@ -38,6 +39,41 @@ } +def _stop_command(process: subprocess.Popen) -> None: + """Stop a timed-out command without draining pipes held by descendants.""" + if os.name == "posix": + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + # GNU timeout and job control can move children to another process + # group in the command's session. On Linux, stop those children too. + proc = Path("/proc") + if proc.is_dir(): + for entry in proc.iterdir(): + if not entry.name.isdecimal(): + continue + try: + fields = (entry / "stat").read_text().rpartition(")")[2].split() + if int(fields[3]) == process.pid: + os.kill(int(entry.name), signal.SIGKILL) + except (OSError, ValueError, IndexError): + continue + try: + process.kill() + except ProcessLookupError: + pass + # A detached descendant can still hold either pipe open. Closing our + # readers makes teardown independent of that descendant's lifetime. + for pipe in (process.stdout, process.stderr): + if pipe is not None: + pipe.close() + try: + process.wait(timeout=1) + except subprocess.TimeoutExpired: + pass + + def run_bash(command: str, *, timeout_seconds: float | None = None) -> tuple[str, bool]: """Run ``command`` with ``bash -c`` and return ``(output, is_error)``. @@ -50,8 +86,9 @@ def run_bash(command: str, *, timeout_seconds: float | None = None) -> tuple[str user's keystrokes. Background children inherit output pipes unless redirected. On POSIX, - timeouts and interruptions kill the command's process group; a normally - completed command leaves background services available to later tools. + timeouts and interruptions kill the command's process group (and other + members of its session on Linux). A normally completed command leaves + background services available to later tools. Text mode translates ``\\r`` to ``\\n`` (universal newlines). """ timeout = BASH_TIMEOUT_SECONDS if timeout_seconds is None else min( @@ -59,7 +96,7 @@ def run_bash(command: str, *, timeout_seconds: float | None = None) -> tuple[str ) if timeout <= 0: return "[command not started: time budget exhausted]", True - with subprocess.Popen( + process = subprocess.Popen( ["bash", "-c", command], stdout=subprocess.PIPE, stderr=subprocess.PIPE, @@ -67,21 +104,18 @@ def run_bash(command: str, *, timeout_seconds: float | None = None) -> tuple[str errors="replace", stdin=subprocess.DEVNULL, start_new_session=os.name == "posix", - ) as process: - try: - stdout, stderr = process.communicate(timeout=timeout) - except BaseException as exc: - try: - if os.name == "posix": - os.killpg(process.pid, signal.SIGKILL) - else: - process.kill() - except ProcessLookupError: - pass - process.communicate() - if isinstance(exc, subprocess.TimeoutExpired): - return f"[command timed out after {timeout:g} seconds]", True - raise + ) + try: + stdout, stderr = process.communicate(timeout=timeout) + except BaseException as exc: + _stop_command(process) + if isinstance(exc, subprocess.TimeoutExpired): + return f"[command timed out after {timeout:g} seconds]", True + raise + finally: + for pipe in (process.stdout, process.stderr): + if pipe is not None: + pipe.close() parts = [] if stdout := stdout.rstrip("\n"): diff --git a/tests/test_bash_tool.py b/tests/test_bash_tool.py index 7a7b4c4..1f92619 100644 --- a/tests/test_bash_tool.py +++ b/tests/test_bash_tool.py @@ -5,7 +5,11 @@ """ from nanopycodeagent import bash_tool +from nanopycodeagent.deadline import DeadlineExceeded, wall_clock_limit +import pytest import shlex +import shutil +import sys import time @@ -84,3 +88,33 @@ def test_normal_exit_preserves_background_service(tmp_path): assert not is_error time.sleep(0.3) assert target.exists() + + +@pytest.mark.skipif(sys.platform != "linux" or not shutil.which("timeout"), + reason="requires Linux and GNU timeout") +@pytest.mark.parametrize("deadline", [False, True]) +def test_timeout_stops_children_in_another_process_group(tmp_path, deadline): + target = tmp_path / "must-not-exist" + child = f"sleep 0.8; touch {shlex.quote(str(target))}" + command = f"timeout 5 bash -c {shlex.quote(child)}; true" + started = time.monotonic() + if deadline: + with pytest.raises(DeadlineExceeded), wall_clock_limit(0.1): + bash_tool.run_bash(command) + else: + output, is_error = bash_tool.run_bash(command, timeout_seconds=0.1) + assert is_error and "timed out" in output + assert time.monotonic() - started < 0.6 + time.sleep(0.9) + assert not target.exists() + + +@pytest.mark.skipif(sys.platform != "linux" or not shutil.which("setsid"), + reason="requires Linux and setsid") +def test_timeout_does_not_drain_pipes_held_by_detached_child(): + started = time.monotonic() + output, is_error = bash_tool.run_bash( + "setsid bash -c 'sleep 0.8' & wait", timeout_seconds=0.1, + ) + assert is_error and "timed out" in output + assert time.monotonic() - started < 0.6 From cdd65ebc99ed837f58467900db7892de82e68842 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Wed, 30 Sep 2026 04:22:08 +0800 Subject: [PATCH 12/12] docs(benchmarks): record fixed-provider repeats and pilot20 results --- .../harbor/reports/dual-budget-20260929.md | 183 +- ...21-dualbudget-fixed-provider-20260929.json | 1670 +++++++++++++++++ 2 files changed, 1837 insertions(+), 16 deletions(-) create mode 100644 benchmarks/harbor/results/tb21-dualbudget-fixed-provider-20260929.json diff --git a/benchmarks/harbor/reports/dual-budget-20260929.md b/benchmarks/harbor/reports/dual-budget-20260929.md index 592e24a..d7305e2 100644 --- a/benchmarks/harbor/reports/dual-budget-20260929.md +++ b/benchmarks/harbor/reports/dual-budget-20260929.md @@ -1,5 +1,150 @@ # Dual-budget finalization with a fixed provider +The repaired candidate (`380271dfd`) passed all six targeted trials and ended +as native `completed` in all six. The control passed five of six and ended as +native `completed` in four. The registered gate passed. The subsequent pilot20 +scored **18/20**, with all 20 runs ending as native `completed`. Caffe accuracy +and PyTorch pipeline numerical correctness failed their independent verifiers; +normal finalization did not imply task success. + +The complete configuration, metrics, and all retained trial rows are in the +[public results](../results/tb21-dualbudget-fixed-provider-20260929.json), including +the four superseded treatment trials. Raw evidence remains in the recorded +Git-ignored `jobs/` directories. + +## Targeted results + +Each row contains all three repetitions of one task and arm. Time covers the +agent process, including post-task cost reconciliation; work time is measured +from the native journal's run start to its last model/tool completion. Costs +include supplementary receipt reconciliation. Native journals and trajectories +retain their original pending cost entries rather than being rewritten. + +| Task | Arm | Reward 1 | Native completed | Replies by repetition | Mean work (min) | Mean total (min) | Mean cost (USD) | +| --- | --- | --- | --- | --- | ---: | ---: | ---: | +| caffe-cifar-10 | Control | 3/3 | 2/3 | 51, 100, 43 | 31.30 | 33.12 | 0.078184 | +| caffe-cifar-10 | Repaired | 3/3 | 3/3 | 51, 72, 59 | 32.76 | 33.27 | 0.057473 | +| mteb-leaderboard | Control | 2/3 | 2/3 | 72, 100, 43 | 19.44 | 21.67 | 0.073795 | +| mteb-leaderboard | Repaired | 3/3 | 3/3 | 71, 91, 59 | 11.48 | 11.99 | 0.083837 | + +The six repaired trials cost $0.423931464 versus $0.455935140 for the six +controls (about 7.0% less). Caffe's mean cost fell 26.5%, while mean total time +was essentially unchanged (+0.4%); mean work time increased 4.7%. MTEB's mean +cost rose 13.6%, while mean total time fell 44.7% and work time fell 41.0%. +These are observations from three repetitions per task, not general effect +estimates or proof that any single component caused the differences. + +Both control tasks in repetition 2 exhausted 100 replies. Caffe still scored +1, but MTEB scored 0 because the required `result.txt` did not exist. In the +repaired MTEB repetition 2, the turn-91 reminder fired with 47:34 of time still +available; that reply ended the run normally and the task passed. The other +five repaired trials finished before either finalization threshold fired. + +All twelve trials have matching pinned task refs, valid native ATIF, and +complete provider/cost audits. Every recorded model POST applied the provider +restriction, and every received generation was attributed to Parasail. One +control connection-establishment timeout was recovered by the SDK before +response headers; it is recorded separately from received generations. No +repaired trial needed a supervisor deadline, forced kill, or reconstructed +trajectory. The longest completed tool call in the repaired cohort was about +120.11 seconds, versus the superseded treatment's observed 600-second cleanup +wait. + +The initial treatment's four completed trials all passed and ended normally, +but they remain separate superseded evidence because of the cleanup defect. +Their total cost was $0.278459916. The unstarted third repetition was cancelled; +all three repaired repetitions were run afresh. No best-run selection was used. + +## Pilot20 regression + +The original pinned 20-task selection ran once, with two concurrent trials, +the same repaired source, model, and fixed provider. Actual native journals +confirm 100 replies, 65536 output tokens, and task-work budgets of each native +agent timeout minus 180 seconds (12, 27, 37, or 57 minutes). Harbor task, setup, +and verifier timeouts were not overridden. Existing verified QEMU bootstrap +packages and the PyTorch model-recovery verifier package cache were reused; +task instructions, image digests, and verifier logic were unchanged. No task +was retried based on its score. + +All 20 tasks were scored: 18 passed and two scored zero. All 20 ended as native +`completed`, with valid native ATIF, matching task refs and runtime budgets, +and the expected source version `0.8.1.dev43+g380271dfd`. There were no Harbor +exceptions, supervisor deadlines, forced kills, or reconstructed trajectories. +All 558 model POSTs applied the provider restriction and match 558 Parasail +generation receipts. There were no failed model attempts or connection failures +before response headers. Full reconciled model cost was **$1.123469172**. +Harbor job wall time was 1 hour 55 minutes 52 seconds with two concurrent trials; +this includes setup and verification, but excludes the later host receipt audit. +The longest completed tool call was about 120.19 seconds. + +| Task | Reward | Replies | Work (min) | Budget (min) | Cost (USD) | +| --- | ---: | ---: | ---: | ---: | ---: | +| caffe-cifar-10 | 0 | 66 | 54.07 | 57 | 0.091648 | +| circuit-fibsqrt | 1 | 26 | 6.20 | 57 | 0.068076 | +| dna-assembly | 1 | 29 | 7.83 | 27 | 0.134251 | +| kv-store-grpc | 1 | 10 | 4.96 | 12 | 0.004265 | +| llm-inference-batching-scheduler | 1 | 27 | 3.71 | 27 | 0.063599 | +| log-summary-date-ranges | 1 | 7 | 0.26 | 12 | 0.004642 | +| merge-diff-arc-agi-task | 1 | 16 | 0.88 | 12 | 0.016840 | +| model-extraction-relu-logits | 1 | 17 | 1.61 | 12 | 0.020341 | +| mteb-leaderboard | 1 | 49 | 9.59 | 57 | 0.054133 | +| openssl-selfsigned-cert | 1 | 13 | 0.56 | 12 | 0.006882 | +| path-tracing | 1 | 22 | 2.01 | 27 | 0.064753 | +| pypi-server | 1 | 17 | 5.03 | 12 | 0.008618 | +| pytorch-model-recovery | 1 | 10 | 2.73 | 12 | 0.009207 | +| qemu-alpine-ssh | 1 | 43 | 9.30 | 12 | 0.060265 | +| regex-chess | 1 | 55 | 36.41 | 57 | 0.214995 | +| regex-log | 1 | 7 | 1.72 | 12 | 0.030141 | +| schemelike-metacircular-eval | 1 | 37 | 20.70 | 37 | 0.082982 | +| torch-pipeline-parallelism | 0 | 44 | 9.78 | 12 | 0.074019 | +| torch-tensor-parallelism | 1 | 44 | 6.86 | 12 | 0.062337 | +| write-compressor | 1 | 19 | 2.77 | 12 | 0.051477 | + +The two failures were task-solution failures with preserved artifacts: + +- **Caffe:** five of six verifier tests passed, including source/build checks, + model/config files, CPU training, and completion of 500 iterations. The + verifier extracted accuracy 0.37 against a strict requirement greater than + 0.45. The agent separately reported full-test accuracy 0.4395 and explicitly + acknowledged missing the requirement. These are distinct measurements, both + below the threshold. Time-based finalization reminders appeared on replies + 64–66, starting with 7:11 remaining; the run ended normally after 66 replies. + All three targeted repaired Caffe trials had passed, so this additional + failure also shows that the targeted result is not a guarantee of reliability. +- **PyTorch pipeline:** two of four verifier tests failed numerical comparisons: + the one-process case differed at `model.norm.fwd` (maximum difference + 5.190534591674805), and the two-process case at `model.layers.1.bwd` + (0.013752378523349762). The agent reported passing its own checks, but the + independent verifier rejected the solution. Time reminders appeared at + replies 43–44, starting with 2:20 remaining, and it ended normally after 44 + replies. The agent's work budget was not exhausted. + +QEMU provides a successful time-based finalization example: reminders started +on reply 40 with 2:57 remaining, and it passed after ending on reply 43. The +other 17 pilot tasks finished before the finalization threshold fired. Together +with the targeted MTEB reply-91 case, these runs exercise both reminder triggers. +They do not establish whether a reminder caused an individual correctness result. + +For historical context, the tracked September 6 pilot scored 8/20 using +`deepseek-v4-flash-0731`, account routing, 50 replies, and 8192 output tokens. +The local September 21 stream-recovery pilot scored 11/20 (six zeros and three +unscored trials), with 50 replies, 65536 tokens, and agent/setup overrides of +3780/3600 seconds. The new 18/20 run changes model, routing, budgets, and agent +source relative to those pilots, so the score difference cannot be attributed +solely to dual-budget finalization. Only the targeted repeated comparison uses +the fixed-provider control described in this report. + +## Decision + +Retain the fixed system prompt, appended budget reminders, early finalization +guidance, and bounded model/tool/cost operations. The registered targeted gate +passed and the full pilot preserved native terminal records for every task. +Keep ordinary defaults at 50 replies and 32768 output tokens; 100/65536 remains +this experiment's configuration. The pilot is not an all-pass result, and this +small study does not establish optimal defaults. Further evaluation should +separate task-solution accuracy and verification quality from runtime +finalization, and use matched controls before attributing score changes. + ## Implementation Headless runs receive append-only reminders of their remaining model replies @@ -15,9 +160,10 @@ using a POSIX main-thread timer. Socket and bash timeouts also use the remaining budget. Bash interruptions terminate the command's process group and, on Linux, other members of its session. Cleanup closes output pipes instead of waiting for detached descendants to close them; reaping the direct child is limited to -one second. Normal exits preserve background services. Stream retries must fit inside the remaining -budget. Cost reconciliation has a separate shared 30-second cap, after which -unresolved costs remain pending and the native Journal/ATIF can be finalized. +one second. Normal exits preserve background services. Stream retries must fit +inside the remaining budget. Cost reconciliation has a separate shared +30-second cap, after which unresolved costs remain pending and the native +Journal/ATIF can be finalized. Callers with an existing real-time alarm, non-main threads, and platforms without `setitimer` receive an explicit error for time-budgeted runs. @@ -28,14 +174,16 @@ without `setitimer` receive an explicit error for time-budgeted runs. that endpoint and `allow_fallbacks=false`. Pinning is applied by the experiment wrapper to both arms, not through a change to account-wide routing settings. - Control: the source/wheel used by experiment 3 (`83e5c064d`). -- Treatment: the committed dual-budget implementation in this branch. +- Treatment: the repaired dual-budget source/wheel (`380271dfd`). - Tasks: the same pinned `caffe-cifar-10` and `mteb-leaderboard` task refs. -- Three attempts per task per arm, in separate preserved jobs, alternating arm - order across pairs. Two simultaneous trials, 100 replies, 65536 output tokens, - 3420 seconds for task work, no whole-task retries. +- Three attempts per task per arm, in separate preserved jobs. The original + order alternated arms; the revision below records the actual order change. + Two simultaneous trials, 100 replies, 65536 output tokens, 3420 seconds for + task work, no whole-task retries. - Native Harbor task/setup/verifier timeouts are preserved. The supervisor remains at 3480 seconds with 120 seconds of finalization grace. -- Dependencies and cached image digests match experiment 3. No fault injection. +- Dependencies and cached image digests match experiment 3. Temperature and + reasoning effort were not explicitly set. No fault injection. - Verify actual provider receipts for all model requests, not just the routing configuration. Preserve incomplete receipts as unknown rather than treating missing costs as zero; post-run reconciliation may supplement benchmark @@ -65,7 +213,8 @@ estimate of this implementation's effect. These small repeated samples can reveal regressions but do not establish a general optimal default. Provider load, cache state, and sampling remain -sources of variation despite a fixed endpoint. +sources of variation despite a fixed endpoint. External task services and +network conditions also remain variable. ## Revision during live trials @@ -73,17 +222,18 @@ The original treatment (`847cc9895`) exposed a subprocess cleanup defect during its second Caffe trial. The model invoked `timeout 600 wget ...`; GNU `timeout` created another process group. Killing the shell's group at 120 seconds left that child holding the output pipes, so the subsequent unbounded `communicate()` -waited for it. This observation blocks pilot20 independently of reward. +waited for it. This observation blocked the original pipeline independently of reward. The repair kills other members of the command's session on Linux, closes the pipe readers without draining them, and bounds direct-child reaping. The first two treatment repetitions remain preserved as superseded evidence; the third was cancelled before launch. The three originally planned control repetitions -remain the comparison baseline. All three treatment repetitions will run again +remained the comparison baseline. All three treatment repetitions were rerun with the repaired, committed wheel and the same provider/configuration. They -will be evaluated together against the same registered gate. This revision -changes the originally alternating execution order; it does not select the -best treatment trials or discard the defect from the report. +were evaluated together against the same registered gate. The actual order was +control 1, superseded treatment 1 and 2, control 2 and 3, then repaired treatment +1, 2, and 3. This changed the originally alternating order without selecting +the best treatment trials or discarding the defect from the report. ## Local validation @@ -91,5 +241,6 @@ The pinned benchmark environment passes 354 tests across the main package and Harbor adapter. Tests cover early turn-based finalization, stable system prompts, reminder journaling, stalled streams, real long-running commands, child-process termination (including GNU `timeout` creating another process -group), bounded cleanup with detached pipe holders, preservation of background services after normal -exit, bounded retries, bounded cost reconciliation, and ATIF compatibility. +group), bounded cleanup with detached pipe holders, preservation of background +services after normal exit, bounded retries, bounded cost reconciliation, and +ATIF compatibility. The PR checks also passed on Python 3.13 and 3.14. diff --git a/benchmarks/harbor/results/tb21-dualbudget-fixed-provider-20260929.json b/benchmarks/harbor/results/tb21-dualbudget-fixed-provider-20260929.json new file mode 100644 index 0000000..ae4b803 --- /dev/null +++ b/benchmarks/harbor/results/tb21-dualbudget-fixed-provider-20260929.json @@ -0,0 +1,1670 @@ +{ + "schema_version": 1, + "generated_at": "2026-09-29T20:19:50.122470+00:00", + "purpose": "Repeated comparison of experiment-3 tail reminders and dual-budget finalization at a fixed provider, followed by a conditional pilot20 regression.", + "control_ref": "83e5c064d4c6e421fc08a90bbd84d2d23b6bb3cf", + "treatment_ref": "380271dfd5e84b4dcc9c7059e71e7ca154c31554", + "model": "openrouter/deepseek/deepseek-v4.1-flash", + "provider_pin": { + "only": [ + "parasail/fp8" + ], + "order": [ + "parasail/fp8" + ], + "allow_fallbacks": false + }, + "dataset": "terminal-bench/terminal-bench-2-1", + "dataset_ref": "sha256:7d7bdc1cbedad549fc1140404bd4dc45e5fd0ea7c4186773687d177ad3a0699a", + "configuration": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420, + "supervisor_seconds": 3480, + "supervisor_grace_seconds": 120, + "native_agent_timeout_seconds": 3600, + "harbor_timeout_overrides": false, + "concurrency": 2, + "attempts_per_task_per_arm": 3, + "automatic_task_retries": 0, + "temperature": "not explicitly set", + "reasoning_effort": "not explicitly set", + "order": [ + "control-1", + "superseded-treatment-1", + "superseded-treatment-2", + "control-2", + "control-3", + "repaired-treatment-1", + "repaired-treatment-2", + "repaired-treatment-3" + ], + "dependencies": [ + "anthropic==1.5.0", + "httpx==0.28.1", + "httpx2==2.12.0", + "pydantic==2.13.5" + ] + }, + "totals": { + "control": { + "finished": 6, + "passed": 5, + "clean_completed": 4, + "known_cost_usd": 0.45593514 + }, + "treatment": { + "finished": 6, + "passed": 6, + "clean_completed": 6, + "known_cost_usd": 0.423931464 + } + }, + "comparisons": [ + { + "task": "terminal-bench/caffe-cifar-10", + "ratios_treatment_over_control": { + "known_cost_usd": 0.7351058084997256, + "duration_seconds": 1.0043512477759022 + }, + "within_gate": true + }, + { + "task": "terminal-bench/mteb-leaderboard", + "ratios_treatment_over_control": { + "known_cost_usd": 1.1360872626590426, + "duration_seconds": 0.5533770549138572 + }, + "within_gate": true + } + ], + "trials": [ + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "control", + "repeat": 1, + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 51, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 3, + "max_tool_seconds": 120.140465318, + "duration_seconds": 1523.705, + "task_work_seconds": 1430.214, + "finalization_seconds": 92.91, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 23077, + "cache_tokens": 1493504, + "output_tokens": 37913, + "known_cost_usd": 0.061379724, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 51 + }, + "missing_receipt_count": 0, + "model_requests": 51, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "control", + "repeat": 1, + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 72, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 1, + "max_tool_seconds": 120.004175785, + "duration_seconds": 1173.561, + "task_work_seconds": 1031.914, + "finalization_seconds": 141.752, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 46144, + "cache_tokens": 2837248, + "output_tokens": 32580, + "known_cost_usd": 0.069962688, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 72 + }, + "missing_receipt_count": 0, + "model_requests": 73, + "all_requests_pinned": true, + "preheader_errors": { + "ConnectTimeout": 1 + }, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "control", + "repeat": 2, + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "max_turns_exhausted", + "model_replies": 100, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 4, + "max_tool_seconds": 120.15632242900001, + "duration_seconds": 2923.15, + "task_work_seconds": 2750.078, + "finalization_seconds": 172.483, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 35761, + "cache_tokens": 4769536, + "output_tokens": 74230, + "known_cost_usd": 0.128421516, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 100 + }, + "missing_receipt_count": 0, + "model_requests": 100, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "control", + "repeat": 2, + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "task_ref_matches": true, + "reward": 0.0, + "exception": null, + "outcome": "max_turns_exhausted", + "model_replies": 100, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 93.176914919, + "duration_seconds": 1597.837, + "task_work_seconds": 1403.128, + "finalization_seconds": 193.858, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 133221, + "cache_tokens": 7274368, + "output_tokens": 32578, + "known_cost_usd": 0.122706108, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 100 + }, + "missing_receipt_count": 0, + "model_requests": 100, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "control", + "repeat": 3, + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 43, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 2, + "max_tool_seconds": 120.100280734, + "duration_seconds": 1515.113, + "task_work_seconds": 1452.949, + "finalization_seconds": 61.557, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 37743, + "cache_tokens": 789760, + "output_tokens": 23907, + "known_cost_usd": 0.04474986, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 43 + }, + "missing_receipt_count": 0, + "model_requests": 43, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "control", + "repeat": 3, + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 43, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 3, + "max_tool_seconds": 120.082351934, + "duration_seconds": 1128.583, + "task_work_seconds": 1064.954, + "finalization_seconds": 63.058, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 25413, + "cache_tokens": 845824, + "output_tokens": 13347, + "known_cost_usd": 0.028715244, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 43 + }, + "missing_receipt_count": 0, + "model_requests": 43, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 1, + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 51, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 4, + "max_tool_seconds": 120.092338981, + "duration_seconds": 1936.217, + "task_work_seconds": 1905.603, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 25910, + "cache_tokens": 1165184, + "output_tokens": 24307, + "known_cost_usd": 0.043932504, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 51 + }, + "missing_receipt_count": 0, + "model_requests": 51, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 1, + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 71, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 3, + "max_tool_seconds": 120.101742266, + "duration_seconds": 1070.236, + "task_work_seconds": 1039.598, + "finalization_seconds": 30.001, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 59919, + "cache_tokens": 3330560, + "output_tokens": 30781, + "known_cost_usd": 0.07489626, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 71 + }, + "missing_receipt_count": 0, + "model_requests": 71, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 2, + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 72, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 3, + "max_tool_seconds": 120.101950144, + "duration_seconds": 1645.164, + "task_work_seconds": 1614.59, + "finalization_seconds": 30.001, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 34514, + "cache_tokens": 2350336, + "output_tokens": 38205, + "known_cost_usd": 0.070302216, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 72 + }, + "missing_receipt_count": 0, + "model_requests": 72, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 2, + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 91, + "model_failures": 0, + "finalization_reminders": 1, + "tool_timeout_count": 1, + "max_tool_seconds": 120.103049356, + "duration_seconds": 598.699, + "task_work_seconds": 568.054, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 59807, + "cache_tokens": 4523776, + "output_tokens": 52259, + "known_cost_usd": 0.107795556, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 91 + }, + "missing_receipt_count": 0, + "model_requests": 91, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 3, + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 59, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 5, + "max_tool_seconds": 120.10221969300001, + "duration_seconds": 2406.529, + "task_work_seconds": 2375.929, + "finalization_seconds": 30.001, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 26343, + "cache_tokens": 1595776, + "output_tokens": 33923, + "known_cost_usd": 0.058185156, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 59 + }, + "missing_receipt_count": 0, + "model_requests": 59, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 3, + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 59, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 79.079163136, + "duration_seconds": 489.225, + "task_work_seconds": 458.628, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 56997, + "cache_tokens": 2656512, + "output_tokens": 29818, + "known_cost_usd": 0.068819772, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 59 + }, + "missing_receipt_count": 0, + "model_requests": 59, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + } + ], + "pilot20_gate": { + "treatment_passed_and_completed": 6, + "provider_audit_required": true, + "max_mean_cost_ratio": 1.5, + "max_mean_duration_ratio": 1.5, + "passed": true, + "failures": [], + "hold_reason": null + }, + "pipeline_status": "finished", + "pilot20": { + "status": "finished", + "budget_policy": "Native agent timeout minus 180 seconds; supervisor at native timeout minus 120 seconds plus 120 seconds grace.", + "trials": [ + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "task_ref_matches": true, + "reward": 0.0, + "exception": null, + "outcome": "completed", + "model_replies": 66, + "model_failures": 0, + "finalization_reminders": 3, + "tool_timeout_count": 6, + "max_tool_seconds": 120.136294878, + "duration_seconds": 3274.557, + "task_work_seconds": 3243.936, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 26490, + "cache_tokens": 2520960, + "output_tokens": 57146, + "known_cost_usd": 0.09164796, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 66 + }, + "missing_receipt_count": 0, + "model_requests": 66, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/circuit-fibsqrt", + "task_ref": "sha256:9bcffe1054bb33249aa578a9a2a74f3c8cca66b0cb7aa1328233f1d31822aae3", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 26, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 1, + "max_tool_seconds": 120.018486391, + "duration_seconds": 402.777, + "task_work_seconds": 372.012, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 12744, + "cache_tokens": 1131776, + "output_tokens": 47885, + "known_cost_usd": 0.068075856, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 26 + }, + "missing_receipt_count": 0, + "model_requests": 26, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 1620 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/dna-assembly", + "task_ref": "sha256:e41a8e94d86019949b08d3b5f88a85f6d943ba0fd85d5e1d5ebb95cb8f66223f", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 29, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 11.071837908, + "duration_seconds": 500.141, + "task_work_seconds": 469.562, + "finalization_seconds": 30.001, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 34247, + "cache_tokens": 1790976, + "output_tokens": 94359, + "known_cost_usd": 0.134250756, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 29 + }, + "missing_receipt_count": 0, + "model_requests": 29, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/kv-store-grpc", + "task_ref": "sha256:973c5d4c111fb61a344457936f1c36400acd2d9e44389e7b319586fe23a7a307", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 10, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 2, + "max_tool_seconds": 120.101613288, + "duration_seconds": 315.712, + "task_work_seconds": 297.664, + "finalization_seconds": 17.373, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 3212, + "cache_tokens": 34176, + "output_tokens": 2580, + "known_cost_usd": 0.004264656, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 10 + }, + "missing_receipt_count": 0, + "model_requests": 10, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 1620 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/llm-inference-batching-scheduler", + "task_ref": "sha256:a3bf47589118daec5124fe3689b4439802c805b2f4ed9d402a48ae73ca2fea67", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 27, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 25.717569813999997, + "duration_seconds": 253.917, + "task_work_seconds": 222.872, + "finalization_seconds": 30.003, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 27840, + "cache_tokens": 968192, + "output_tokens": 41198, + "known_cost_usd": 0.063598752, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 27 + }, + "missing_receipt_count": 0, + "model_requests": 27, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/log-summary-date-ranges", + "task_ref": "sha256:27b074a2f10fff7606e096f3abd8dced418ad8fda0f53d88acbe477f2d9ceaf6", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 7, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 0.154241255, + "duration_seconds": 29.233, + "task_work_seconds": 15.405, + "finalization_seconds": 13.182, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 5056, + "cache_tokens": 35712, + "output_tokens": 2426, + "known_cost_usd": 0.004642272, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 7 + }, + "missing_receipt_count": 0, + "model_requests": 7, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/merge-diff-arc-agi-task", + "task_ref": "sha256:6aab6511a5344ce87698293bb1ce4cc51d9a45f1ad9f0c075d2a83197b36727d", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 16, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 0.028048651, + "duration_seconds": 75.19, + "task_work_seconds": 52.522, + "finalization_seconds": 22.083, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 9472, + "cache_tokens": 161664, + "output_tokens": 10857, + "known_cost_usd": 0.016839984, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 16 + }, + "missing_receipt_count": 0, + "model_requests": 16, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/model-extraction-relu-logits", + "task_ref": "sha256:1ae5045ad68b5d34c3398b612066a07c4a08b6dc330d28868ec4021e17c94b17", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 17, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 2.92958362, + "duration_seconds": 121.952, + "task_work_seconds": 96.869, + "finalization_seconds": 24.416, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 5495, + "cache_tokens": 212736, + "output_tokens": 14513, + "known_cost_usd": 0.020340516, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 17 + }, + "missing_receipt_count": 0, + "model_requests": 17, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 49, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 2, + "max_tool_seconds": 120.102229569, + "duration_seconds": 606.164, + "task_work_seconds": 575.513, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 49821, + "cache_tokens": 1910656, + "output_tokens": 23102, + "known_cost_usd": 0.054132636, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 49 + }, + "missing_receipt_count": 0, + "model_requests": 49, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/openssl-selfsigned-cert", + "task_ref": "sha256:d4afa2bd2a9ba1420db8d6cfde42ffdb4873ae2d955c35014e8da94444c83302", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 13, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 0.175042284, + "duration_seconds": 52.04, + "task_work_seconds": 33.531, + "finalization_seconds": 17.823, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 3941, + "cache_tokens": 60416, + "output_tokens": 4448, + "known_cost_usd": 0.006882396, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 13 + }, + "missing_receipt_count": 0, + "model_requests": 13, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 1620 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/path-tracing", + "task_ref": "sha256:cf56094c881a488b27e9f204a638a7e78ed7d55e12dc3064108c93357190314c", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 22, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 1.564569333, + "duration_seconds": 151.253, + "task_work_seconds": 120.699, + "finalization_seconds": 30.001, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 70459, + "cache_tokens": 1017984, + "output_tokens": 31256, + "known_cost_usd": 0.064752804, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 22 + }, + "missing_receipt_count": 0, + "model_requests": 22, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/pypi-server", + "task_ref": "sha256:1a1e0542f58e2d3362fec17a9bbb98667717d9a4a3e9a4c8413d3150a4fa0ff1", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 17, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 2, + "max_tool_seconds": 120.090140276, + "duration_seconds": 329.786, + "task_work_seconds": 301.823, + "finalization_seconds": 27.3, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 5378, + "cache_tokens": 93696, + "output_tokens": 5369, + "known_cost_usd": 0.008618376, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 17 + }, + "missing_receipt_count": 0, + "model_requests": 17, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/pytorch-model-recovery", + "task_ref": "sha256:2e628841cff93290919172398e573e794f34c95d9382b7425adde4364022decc", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 10, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 1, + "max_tool_seconds": 120.185438948, + "duration_seconds": 180.994, + "task_work_seconds": 163.612, + "finalization_seconds": 16.628, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 8874, + "cache_tokens": 87936, + "output_tokens": 5014, + "known_cost_usd": 0.009206616, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 10 + }, + "missing_receipt_count": 0, + "model_requests": 10, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/qemu-alpine-ssh", + "task_ref": "sha256:60b7050b0e0aa51641208cf65766743d340e59575db2c4d2f8628240846c2a28", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 43, + "model_failures": 0, + "finalization_reminders": 4, + "tool_timeout_count": 1, + "max_tool_seconds": 120.101231115, + "duration_seconds": 588.692, + "task_work_seconds": 558.011, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 56504, + "cache_tokens": 2307968, + "output_tokens": 24555, + "known_cost_usd": 0.060265008, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 43 + }, + "missing_receipt_count": 0, + "model_requests": 43, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/regex-chess", + "task_ref": "sha256:e763e0ac1c9759081af0a4a82ba51b8cf9ae5485a93de3bbe42d7d344597bd78", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 55, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 4, + "max_tool_seconds": 120.100518331, + "duration_seconds": 2215.128, + "task_work_seconds": 2184.393, + "finalization_seconds": 30.003, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 19038, + "cache_tokens": 6477568, + "output_tokens": 142015, + "known_cost_usd": 0.214994808, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 55 + }, + "missing_receipt_count": 0, + "model_requests": 55, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/regex-log", + "task_ref": "sha256:802c16cfd132e6c457529cb864be5a757c1b23b6cadc57f2d01983cb0110292a", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 7, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 0.8028998940000001, + "duration_seconds": 114.899, + "task_work_seconds": 103.486, + "finalization_seconds": 10.832, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 3260, + "cache_tokens": 131456, + "output_tokens": 23645, + "known_cost_usd": 0.030140736, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 7 + }, + "missing_receipt_count": 0, + "model_requests": 7, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 2220 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/schemelike-metacircular-eval", + "task_ref": "sha256:58130c2166c3115276dc8592f358e326ff2d81ea852e3d88636c82fd1dff57e6", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 37, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 3, + "max_tool_seconds": 120.102095111, + "duration_seconds": 1272.857, + "task_work_seconds": 1242.191, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 20695, + "cache_tokens": 1490304, + "output_tokens": 56526, + "known_cost_usd": 0.082981524, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 37 + }, + "missing_receipt_count": 0, + "model_requests": 37, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/torch-pipeline-parallelism", + "task_ref": "sha256:db605337c749a872cea7b5b413429b3915bb4c3efe0f7875f0c46ce81bd8c4fb", + "task_ref_matches": true, + "reward": 0.0, + "exception": null, + "outcome": "completed", + "model_replies": 44, + "model_failures": 0, + "finalization_reminders": 2, + "tool_timeout_count": 0, + "max_tool_seconds": 42.018512412, + "duration_seconds": 617.145, + "task_work_seconds": 586.557, + "finalization_seconds": 30.001, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 17826, + "cache_tokens": 1736448, + "output_tokens": 48544, + "known_cost_usd": 0.074019288, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 44 + }, + "missing_receipt_count": 0, + "model_requests": 44, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/torch-tensor-parallelism", + "task_ref": "sha256:f32ce74a5aeb6638480247ab799fe46127bbee631acdd0921b0f394ec49b3684", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 44, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 45.866061207, + "duration_seconds": 442.083, + "task_work_seconds": 411.479, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 14793, + "cache_tokens": 1427328, + "output_tokens": 41113, + "known_cost_usd": 0.062337468, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 44 + }, + "missing_receipt_count": 0, + "model_requests": 44, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 720 + }, + "arm": "pilot20", + "repeat": 1, + "task": "terminal-bench/write-compressor", + "task_ref": "sha256:d9ddd9a8e925e2c566b37b2492cbf995afecefe58874e4043ef78d7f3c892c7e", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 19, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 0.130977659, + "duration_seconds": 194.699, + "task_work_seconds": 166.313, + "finalization_seconds": 27.743, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 6762, + "cache_tokens": 543360, + "output_tokens": 38490, + "known_cost_usd": 0.05147676, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 19 + }, + "missing_receipt_count": 0, + "model_requests": 19, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + } + ], + "summary": { + "planned": 20, + "finished": 20, + "passed": 18, + "scored_zero": 2, + "unscored": 0, + "native_completed": 20, + "known_cost_usd": 1.123469172, + "all_costs_complete": true, + "all_providers_audited": true, + "all_task_refs_match": true, + "all_native_atif_valid": true, + "supervisor_deadlines": 0, + "forced_kills": 0, + "model_replies": 558, + "model_failures": 0, + "outcomes": { + "completed": 20 + }, + "exceptions": {}, + "runtime_budgets_match": true, + "selection_matches": true + }, + "configuration": { + "attempts_per_task": 1, + "concurrency": 2, + "max_turns": 100, + "max_tokens": 65536, + "native_agent_timeouts_seconds": { + "write-compressor": 900, + "torch-tensor-parallelism": 900, + "schemelike-metacircular-eval": 2400, + "kv-store-grpc": 900, + "pypi-server": 900, + "dna-assembly": 1800, + "torch-pipeline-parallelism": 900, + "qemu-alpine-ssh": 900, + "openssl-selfsigned-cert": 900, + "regex-chess": 3600, + "log-summary-date-ranges": 900, + "model-extraction-relu-logits": 900, + "path-tracing": 1800, + "regex-log": 900, + "caffe-cifar-10": 3600, + "mteb-leaderboard": 3600, + "llm-inference-batching-scheduler": 1800, + "pytorch-model-recovery": 900, + "circuit-fibsqrt": 3600, + "merge-diff-arc-agi-task": 900 + }, + "harbor_timeout_overrides": false, + "automatic_task_retries": 0 + }, + "job_started_at": "2026-09-29T18:22:14.174595+00:00", + "job_finished_at": "2026-09-29T20:18:05.992174+00:00", + "job_wall_seconds": 6951.817579, + "agent_versions": [ + "0.8.1.dev43+g380271dfd" + ], + "failure_analysis": { + "terminal-bench/torch-pipeline-parallelism": { + "classification": "task_solution_validation", + "observation": "Two verifier tests passed and two failed. World size 1 failed a model.norm forward-value comparison (maximum difference 5.190534591674805); world size 2 failed a model.layers.1 backward-value comparison (maximum difference 0.013752378523349762).", + "runtime": "The native run completed after 44 replies. The first time-based finalization reminder occurred at reply 43 with 2:20 of the 12-minute work budget remaining. No agent deadline or supervisor kill occurred.", + "limit": "The agent reported passing its own numerical checks, but the independent verifier rejected the solution. A single run cannot isolate whether the finalization reminder affected correctness.", + "evidence": "jobs/tb21-dualbudget-repaired-pilot20-20260929/torch-pipeline-parallelism__ednH42s/verifier/test-stdout.txt" + }, + "terminal-bench/caffe-cifar-10": { + "classification": "task_solution_accuracy", + "observation": "Five verifier tests passed and one failed. The verifier extracted test accuracy 0.37 against a strict threshold greater than 0.45. Source/build, model and prototxt artifacts, CPU configuration, and completion of 500 training iterations passed.", + "runtime": "The native run completed after 66 replies, with three time-based finalization reminders starting at reply 64 with 7:11 of the 57-minute work budget remaining. No hard deadline, supervisor termination, or reconstructed trajectory was needed.", + "agent_report": "The final answer explicitly acknowledged missing the required accuracy. It reported its own full-test accuracy as 0.4395, also below the threshold; this is distinct from the verifier-extracted 0.37.", + "limit": "All three targeted repaired Caffe trials passed, but this additional pilot trial did not. Artifact preservation and normal finalization do not guarantee task correctness; a single trial cannot isolate the effect of finalization guidance or external data-download conditions.", + "evidence": "jobs/tb21-dualbudget-repaired-pilot20-20260929/caffe-cifar-10__ywdMQmN/verifier/test-stdout.txt" + } + } + }, + "comparison_limits": "Three attempts per task per arm are diagnostic, not a general pass-rate estimate. Provider endpoint is fixed, but sampling, cache state, provider load, backend and external task services/network conditions remain variable. Historical pilot20 used a different model, routing, and timeout configuration; the new pilot20 is a regression run, not a controlled attribution. Connection-establishment failures are recorded separately from received generation receipts.", + "artifacts": { + "suite": "jobs/tb21-dualbudget-repaired-20260929-record", + "note": "Raw trials, request metadata, generation IDs and supplementary receipts remain in Git-ignored jobs/. Native journals and trajectories are not rewritten by supplementary accounting." + }, + "revision": "Repair discovered during original treatment repetition 2. Preserve the first two original treatment repetitions as superseded evidence; cancel the unstarted third. Reuse all three original controls and rerun every treatment repetition; no trial selection.", + "superseded_treatment": { + "source_ref": "847cc98956562b484897780ba662ee9f42df2cd3", + "hold_reason": "Live treatment r2 exposed unbounded pipe draining after GNU timeout moved children into another process group. The treatment must be repaired and re-evaluated before pilot20.", + "cancelled_before_start": [ + "treatment-3" + ], + "trials": [ + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 1, + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 56, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 5, + "max_tool_seconds": 120.070739279, + "duration_seconds": 2092.863, + "task_work_seconds": 2060.808, + "finalization_seconds": 30.001, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 35683, + "cache_tokens": 1860480, + "output_tokens": 43643, + "known_cost_usd": 0.07423938, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 56 + }, + "missing_receipt_count": 0, + "model_requests": 56, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 1, + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 86, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 2, + "max_tool_seconds": 120.100472566, + "duration_seconds": 1065.016, + "task_work_seconds": 1032.916, + "finalization_seconds": 30.002, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 57835, + "cache_tokens": 3816576, + "output_tokens": 39143, + "known_cost_usd": 0.087221556, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 86 + }, + "missing_receipt_count": 0, + "model_requests": 86, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 2, + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 92, + "model_failures": 0, + "finalization_reminders": 2, + "tool_timeout_count": 3, + "max_tool_seconds": 600.003733743, + "duration_seconds": 2274.45, + "task_work_seconds": 2244.374, + "finalization_seconds": 30.001, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 38898, + "cache_tokens": 3796096, + "output_tokens": 56337, + "known_cost_usd": 0.102050376, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 92 + }, + "missing_receipt_count": 0, + "model_requests": 93, + "all_requests_pinned": true, + "preheader_errors": { + "ConnectError": 1 + }, + "unattributed_request_count": 0 + }, + { + "runtime_limits": { + "max_turns": 100, + "max_tokens": 65536, + "time_budget_seconds": 3420 + }, + "arm": "treatment", + "repeat": 2, + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "task_ref_matches": true, + "reward": 1.0, + "exception": null, + "outcome": "completed", + "model_replies": 19, + "model_failures": 0, + "finalization_reminders": 0, + "tool_timeout_count": 0, + "max_tool_seconds": 8.509400348, + "duration_seconds": 151.81, + "task_work_seconds": 120.272, + "finalization_seconds": 30.001, + "deadline_reached": false, + "forced_kill": false, + "trajectory_source": "native", + "atif_valid": true, + "input_tokens": 21509, + "cache_tokens": 313984, + "output_tokens": 5510, + "known_cost_usd": 0.014948604, + "cost_complete": true, + "provider_audit_ok": true, + "providers": { + "Parasail": 19 + }, + "missing_receipt_count": 0, + "model_requests": 19, + "all_requests_pinned": true, + "preheader_errors": {}, + "unattributed_request_count": 0 + } + ] + } +}