diff --git a/benchmarks/harbor/results/tb21-v41flash-6task-20260925.json b/benchmarks/harbor/results/tb21-v41flash-6task-20260925.json new file mode 100644 index 0000000..96fd39c --- /dev/null +++ b/benchmarks/harbor/results/tb21-v41flash-6task-20260925.json @@ -0,0 +1,125 @@ +{ + "schema_version": 1, + "generated_at": "2026-09-26T13:24:25.932705+00:00", + "purpose": "Re-run the six tasks that scored reward 0 in tb21-stream-recovery-pilot20-20260921 with a different model (deepseek/deepseek-v4.1-flash) to see whether they were model-limited.", + "agent_ref": "1386df48ef9c84d7b65c33fcf0f903fad7036a79", + "agent_version": "0.8.1.dev30+g1386df48e", + "model": "openrouter/deepseek/deepseek-v4.1-flash", + "dataset": "terminal-bench/terminal-bench-2-1", + "dataset_ref": "sha256:7d7bdc1cbedad549fc1140404bd4dc45e5fd0ea7c4186773687d177ad3a0699a", + "selection": "The six tasks with reward 0 in the 2026-09-21 stream-recovery 20-task run, with their pinned task refs.", + "configuration": { + "max_tokens": 65536, + "project_default_max_tokens": 32768, + "max_turns": 50, + "agent_execution_limit_seconds": 3600, + "finalization_grace_seconds": 120, + "harbor_outer_limit_seconds": 3780, + "agent_setup_limit_seconds": 3600, + "maximum_simultaneous_trials": 2, + "automatic_harbor_retries": 0, + "attempts_per_task": 1, + "verifier_limits": "Native task deadlines", + "dependency_constraints": [ + "anthropic==1.5.0", + "httpx==0.28.1", + "httpx2==2.12.0", + "pydantic==2.13.5" + ] + }, + "baseline": { + "job": "tb21-stream-recovery-pilot20-20260921", + "model": "openrouter/deepseek/deepseek-v4-flash-0731", + "note": "All six selected tasks scored reward 0 in that run." + }, + "totals": { + "passed": 4, + "scored_zero": 2, + "unscored": 0, + "model_attempts": 206, + "input_tokens": 8122046, + "cache_tokens": 7444992, + "output_tokens": 219443, + "cost_usd": 0.278759041 + }, + "trials": [ + { + "task": "terminal-bench/kv-store-grpc", + "task_ref": "sha256:973c5d4c111fb61a344457936f1c36400acd2d9e44389e7b319586fe23a7a307", + "reward": 1.0, + "outcome": "completed", + "model_replies": 17, + "input_tokens": 128200, + "cache_tokens": 113024, + "output_tokens": 7949, + "cost_usd": 0.008440065, + "tool_calls": 18 + }, + { + "task": "terminal-bench/dna-assembly", + "task_ref": "sha256:e41a8e94d86019949b08d3b5f88a85f6d943ba0fd85d5e1d5ebb95cb8f66223f", + "reward": 1.0, + "outcome": "completed", + "model_replies": 34, + "input_tokens": 1694676, + "cache_tokens": 1604352, + "output_tokens": 73976, + "cost_usd": 0.069021966, + "tool_calls": 33 + }, + { + "task": "terminal-bench/log-summary-date-ranges", + "task_ref": "sha256:27b074a2f10fff7606e096f3abd8dced418ad8fda0f53d88acbe477f2d9ceaf6", + "reward": 1.0, + "outcome": "completed", + "model_replies": 6, + "input_tokens": 45656, + "cache_tokens": 36992, + "output_tokens": 2315, + "cost_usd": 0.005224672, + "tool_calls": 8 + }, + { + "task": "terminal-bench/circuit-fibsqrt", + "task_ref": "sha256:9bcffe1054bb33249aa578a9a2a74f3c8cca66b0cb7aa1328233f1d31822aae3", + "reward": 1.0, + "outcome": "completed", + "model_replies": 49, + "input_tokens": 2932772, + "cache_tokens": 2639232, + "output_tokens": 64148, + "cost_usd": 0.080010864, + "tool_calls": 51 + }, + { + "task": "terminal-bench/caffe-cifar-10", + "task_ref": "sha256:7b0045106d7d5af724efe96b610ba64f7893f5c88528401c573c4d47e384e2bf", + "reward": 0.0, + "outcome": "max_turns_exhausted", + "model_replies": 50, + "input_tokens": 1525280, + "cache_tokens": 1385344, + "output_tokens": 41545, + "cost_usd": 0.065742942, + "tool_calls": 71 + }, + { + "task": "terminal-bench/mteb-leaderboard", + "task_ref": "sha256:484f6d7008a05b5b8640fc6618a384b8c9447cd76f85416c8a595028d29bff9c", + "reward": 0.0, + "outcome": "max_turns_exhausted", + "model_replies": 50, + "input_tokens": 1795462, + "cache_tokens": 1666048, + "output_tokens": 29510, + "cost_usd": 0.050318532, + "tool_calls": 76 + } + ], + "superseded_run": { + "job": "tb21-v4flash-6task-20260925", + "model": "openrouter/deepseek/deepseek-v4-flash", + "status": "aborted by the operator before completion; not an experiment result." + }, + "comparison_limits": "Single attempt per task; routing, sampling, cache and provider backend are uncontrolled. This is a one-off model comparison, not a controlled estimate of a model-ability difference." +} diff --git a/docs/dev_notes/en/0.8.x.md b/docs/dev_notes/en/0.8.x.md index 9ae1836..6057f38 100644 --- a/docs/dev_notes/en/0.8.x.md +++ b/docs/dev_notes/en/0.8.x.md @@ -1074,3 +1074,60 @@ The reconciled total covers all 267 recorded model replies. This shows that boun The proposed direction is to turn this one-off follow-up into a reusable supported capability that integrates with evaluation and can run independently after tasks finish. Read unresolved generation IDs, query later, preserve receipts, and produce updated aggregates. Preserve original results, avoid counting any charge twice, and distinguish known subtotals from complete totals. Unresolved amounts remain unknown rather than zero, and reconciliation failures must not change task scores. Whether this becomes a separate command or an automatic post-benchmark stage is undecided. The goal is to continue reconciliation beyond task finalization without delaying agent exit indefinitely while waiting for billing records. + +### Re-running the six failed tasks with a different model + +**Run on 2026-09-25.** In the previous stream-recovery retest, six of the twenty tasks scored reward 0 under `deepseek/deepseek-v4-flash-0731`: `kv-store-grpc`, `dna-assembly`, `log-summary-date-ranges`, `caffe-cifar-10`, `mteb-leaderboard`, and `circuit-fibsqrt`. To judge whether those failures were model-limited, this round re-ran exactly those tasks under the same conditions, changing only the model. This is not a controlled experiment: each task ran once, and routing, sampling, cache, and provider backend were not held fixed. + +**Model choice.** The goal was to switch to a DeepSeek Flash model. OpenRouter has no `deepseek-flash` slug, and aliases such as `deepseek/deepseek-v4-flash-latest` are rejected as invalid model IDs by the Anthropic Messages endpoint. The run therefore used **`deepseek/deepseek-v4.1-flash`** (V4.1 Flash), after a single minimal Messages request confirmed it works and supports tool use. An earlier attempt with `deepseek/deepseek-v4-flash` (V4 Flash 0423) was aborted before it finished; its artifacts are kept under `jobs/tb21-v4flash-6task-20260925/` and are not treated as an experiment result. Only the V4.1 Flash round is recorded here. + +**Fixed conditions.** + +| Item | This round | +| --- | --- | +| Job | `tb21-v41flash-6task-20260925` | +| Agent source | `main` commit `1386df48ef9c84d7b65c33fcf0f903fad7036a79`; wheel `0.8.1.dev30+g1386df48e` | +| Model | `openrouter/deepseek/deepseek-v4.1-flash` | +| Dataset | `terminal-bench/terminal-bench-2-1`, `sha256:7d7bdc1c…a0699a` | +| Tasks | The six above, each with its pinned task ref | +| Generation / turns | At most 65536 tokens per reply; at most 50 complete replies per task | +| Deadlines | Agent 3600 s, finalization grace 120 s, Harbor outer limit 3780 s, setup 3600 s; verifiers keep their native task deadlines | +| Concurrency / retries | Concurrency 2; one attempt per task, zero automatic Harbor retries | +| Dependencies | Anthropic SDK 1.5.0, HTTPX 0.28.1, HTTPX2 2.12.0, Pydantic 2.13.5 | +| Images | Reused the cached baseline images, with RepoDigests checked before the run | + +**Results.** All six tasks finished and all received an official score: **four passed and two failed**. + +| Metric | This round | +| --- | --- | +| Passed / failed / unscored | **4 / 2 / 0** | +| Journal model calls | **206 started, 206 returned complete** | +| `model.failed` / HTTP response-body interruptions | **0 / 0** | +| Input / cache / output tokens | 8,122,046 / 7,444,992 / 219,443 | +| Model cost | **$0.278759041** (all per-call receipts complete) | +| Infrastructure exceptions | 0 | + +| Task | Reward | Outcome | Model replies | Cost (USD) | +| --- | ---: | --- | ---: | ---: | +| `kv-store-grpc` | 1 | `completed` | 17 | 0.008440065 | +| `dna-assembly` | 1 | `completed` | 34 | 0.069021966 | +| `log-summary-date-ranges` | 1 | `completed` | 6 | 0.005224672 | +| `circuit-fibsqrt` | 1 | `completed` | 49 | 0.080010864 | +| `caffe-cifar-10` | 0 | `max_turns_exhausted` | 50 | 0.065742942 | +| `mteb-leaderboard` | 0 | `max_turns_exhausted` | 50 | 0.050318532 | + +**The two remaining failures were both turn exhaustion, not crashes or transport faults.** + +- `caffe-cifar-10`: CPU training was too slow; within 50 turns it reached only `Iteration 200`, so `cifar10_quick_iter_500.caffemodel` and `Iteration 500` never existed. The official suite passed 2 of 6 tests and failed 3 (one more passed). The bottleneck is the turn limit combined with training time. +- `mteb-leaderboard`: `/app/result.txt` was never produced and both official tests failed. The model kept reading MTEB source and ran out of turns without writing the result file. Every historical run of this task has failed the same way. + +**Comparison with history.** All six tasks scored reward 0 in the 2026-09-21 run under 0731. Under V4.1 Flash: + +- `kv-store-grpc` and `log-summary-date-ranges` passed again (they had also passed in the earlier 8192 and 09-14/15 runs, so this is a recovery rather than a first pass). +- `dna-assembly` **passed for the first time**, with a clean `completed`; every prior attempt was either 0 or unscored. +- `circuit-fibsqrt` **passed cleanly for the first time**; its two earlier reward-1 results were both accompanied by an execution limit or trajectory reconstruction anomaly. +- `caffe-cifar-10` and `mteb-leaderboard` still failed, and have never passed. + +**Evidence boundary.** Routing, sampling, cache, and backend were not controlled while the model changed, and each task ran only once, so **0/6 → 4/6 cannot be attributed to the model difference**; it is only a signal to verify later. + +**Artifacts.** Raw trials are under `jobs/tb21-v41flash-6task-20260925/`, with the experiment record under `jobs/tb21-v41flash-6task-20260925-record/` (`manifest.json`, `summary.json`, `report.md`); these live in Git-ignored `jobs/` and were not uploaded. Version control stores the structured result [`tb21-v41flash-6task-20260925.json`](../../../benchmarks/harbor/results/tb21-v41flash-6task-20260925.json) and this summary. diff --git a/docs/dev_notes/zh-CN/0.8.x.md b/docs/dev_notes/zh-CN/0.8.x.md index 9cea304..7cfe945 100644 --- a/docs/dev_notes/zh-CN/0.8.x.md +++ b/docs/dev_notes/zh-CN/0.8.x.md @@ -1465,3 +1465,81 @@ generation ID,稍后重新查询,保存补查回执,再生成更新后的 具体采用独立命令还是批量评测后的自动步骤,尚未确定。重点是让补查可以跨越任务 收尾阶段继续进行,避免为了等待账单而长时间拖住 agent 退出。 + +### 更换模型复跑未通过的 6 题 + +**2026-09-25 实跑。** 上一轮流中断恢复复测在 `deepseek/deepseek-v4-flash-0731` +下,20 题里有 6 题 reward 为 0:`kv-store-grpc`、`dna-assembly`、 +`log-summary-date-ranges`、`caffe-cifar-10`、`mteb-leaderboard`、`circuit-fibsqrt`。 +为了判断这些失败是否受模型限制,本轮用同一批题、同一套运行条件,只更换模型把它们 +重跑一遍。这不是受控实验:每题只跑 1 次,路由、采样、缓存和服务端后端都没有固定。 + +**模型选择。** 目标是换一个 DeepSeek Flash 模型。OpenRouter 上并没有 +`deepseek-flash` 这个 slug;`deepseek/deepseek-v4-flash-latest` 等别名会被 +Anthropic Messages 接口判为无效模型。最终使用 **`deepseek/deepseek-v4.1-flash`** +(V4.1 Flash),并先发一次最小 Messages 请求确认它可用且支持工具调用。此前还曾用 +`deepseek/deepseek-v4-flash`(V4 Flash 0423)启动过一轮,但在完成前被中止,产物 +保留在 `jobs/tb21-v4flash-6task-20260925/`,不作为实验结果;本文只记录 V4.1 Flash +这轮。 + +**固定条件。** + +| 项目 | 本轮配置 | +| --- | --- | +| Job | `tb21-v41flash-6task-20260925` | +| Agent 源码 | `main` 提交 `1386df48ef9c84d7b65c33fcf0f903fad7036a79`;wheel `0.8.1.dev30+g1386df48e` | +| Model | `openrouter/deepseek/deepseek-v4.1-flash` | +| Dataset | `terminal-bench/terminal-bench-2-1`,`sha256:7d7bdc1c…a0699a` | +| 题目 | 上述 6 题,沿用各自固定 task ref | +| 生成/轮数 | 每次回复最多 65536 tokens;每题最多 50 次完整回复 | +| 时限 | agent 3600 秒、收尾宽限 120 秒、Harbor 外层 3780 秒、安装 3600 秒;verifier 保留题目原始时限 | +| 并发/重试 | 并发 2;每题 1 次尝试,Harbor 自动重试 0 | +| 依赖 | Anthropic SDK 1.5.0、HTTPX 0.28.1、HTTPX2 2.12.0、Pydantic 2.13.5 | +| 镜像 | 沿用基线缓存镜像,运行前核对 RepoDigests | + +**结果。** 6 题全部结束、全部得到官方评分,**通过 4 题、未通过 2 题**。 + +| 指标 | 本轮结果 | +| --- | --- | +| 通过/未通过/未评分 | **4 / 2 / 0** | +| Journal 模型调用 | **206 次启动,206 次完整返回** | +| `model.failed` / HTTP 响应体中断 | **0 / 0** | +| input/cache/output tokens | 8,122,046 / 7,444,992 / 219,443 | +| 模型费用 | **$0.278759041**(逐次回执完整) | +| 基础设施异常 | 0 | + +| 题目 | Reward | 终态 | 模型回复 | 费用(USD) | +| --- | ---: | --- | ---: | ---: | +| `kv-store-grpc` | 1 | `completed` | 17 | 0.008440065 | +| `dna-assembly` | 1 | `completed` | 34 | 0.069021966 | +| `log-summary-date-ranges` | 1 | `completed` | 6 | 0.005224672 | +| `circuit-fibsqrt` | 1 | `completed` | 49 | 0.080010864 | +| `caffe-cifar-10` | 0 | `max_turns_exhausted` | 50 | 0.065742942 | +| `mteb-leaderboard` | 0 | `max_turns_exhausted` | 50 | 0.050318532 | + +**两题仍未通过,原因都是 50 轮耗尽,不是崩溃或传输故障。** + +- `caffe-cifar-10`:CPU 训练太慢,50 轮内只走到 `Iteration 200`, + `cifar10_quick_iter_500.caffemodel` 与 `Iteration 500` 都不存在;官方 6 项测试 + 通过 2 项、失败 3 项(另有 1 项通过)。瓶颈是轮数上限加训练耗时。 +- `mteb-leaderboard`:始终没有生成 `/app/result.txt`,官方 2 项测试全部失败; + 模型一直在阅读 MTEB 源码,50 轮耗尽仍未写出结果文件。这题在每一轮历史实验里 + 都是同一种死法。 + +**与历史的对比。** 这 6 题在 2026-09-21 的 0731 运行中全部 reward 0;换到 +V4.1 Flash 后: + +- `kv-store-grpc`、`log-summary-date-ranges` 恢复通过(它们在更早的 8192 与 + 09-14~15 运行中也曾通过,属于回归恢复)。 +- `dna-assembly` **首次通过**,且是干净的 `completed`;此前所有尝试都是 0 或未评分。 +- `circuit-fibsqrt` **首次干净通过**;此前两次 reward 1 都伴随执行上限或轨迹重建异常。 +- `caffe-cifar-10`、`mteb-leaderboard` 仍未通过,且历史上从未通过。 + +**证据边界。** 换模型的同时没有控制路由、采样、缓存和后端,每题也只跑了 1 次, +因此 **0/6 → 4/6 不能直接归因于模型差异**,只能作为一个待验证的信号。 + +**产物位置。** 原始 trial 在 `jobs/tb21-v41flash-6task-20260925/`,实验记录目录为 +`jobs/tb21-v41flash-6task-20260925-record/`(`manifest.json`、`summary.json`、 +`report.md`);这些位于 Git 忽略的 `jobs/`,未上传。版本管理保存结构化结果 +[`tb21-v41flash-6task-20260925.json`](../../../benchmarks/harbor/results/tb21-v41flash-6task-20260925.json) +与本节汇总。