From 9fce681f9e5a467b1bcf478a2c90c00ffc83b956 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Thu, 1 Oct 2026 20:29:02 +0800 Subject: [PATCH 1/3] feat(cost): estimate DeepSeek cost from token usage DeepSeek's official Anthropic-compatible endpoint returns token usage but no per-request cost, no generation id, and no usage or billing endpoint, so the existing OpenRouter usage/generation reconciliation cannot resolve a cost. Add estimated_cost(), which prices the returned input, output, and cache token counts with a static DeepSeek price list. It is used only after usage_cost() fails, so a provider-reported cost always wins, and it is marked kind: estimated to stay distinct from a reported or reconciled cost. --- src/nanopycodeagent/agent.py | 5 +++ src/nanopycodeagent/cost.py | 60 ++++++++++++++++++++++++++++++++++++ tests/test_cost.py | 47 ++++++++++++++++++++++++++++ 3 files changed, 112 insertions(+) diff --git a/src/nanopycodeagent/agent.py b/src/nanopycodeagent/agent.py index cb4ee32..ced638e 100644 --- a/src/nanopycodeagent/agent.py +++ b/src/nanopycodeagent/agent.py @@ -42,6 +42,7 @@ from .atif import project_atif, write_atif from .bash_tool import run_bash from .cost import ( + estimated_cost, pending_cost, resolve_generation_cost, usage_cost, @@ -767,6 +768,10 @@ def _run_model_loop( ), "generation_id": generation_id, "cost": usage_cost(usage if isinstance(usage, dict) else None) + or estimated_cost( + str(getattr(message, "model", None) or model), + usage if isinstance(usage, dict) else None, + ) or pending_cost(generation_id), "duration_ms": (model_completed_ns - model_started_ns) / 1_000_000, "source_timestamp": utc_now(), diff --git a/src/nanopycodeagent/cost.py b/src/nanopycodeagent/cost.py index ea54ffb..a30d7eb 100644 --- a/src/nanopycodeagent/cost.py +++ b/src/nanopycodeagent/cost.py @@ -15,6 +15,66 @@ DEFAULT_RETRY_DELAYS = (1.0, 2.0, 4.0, 8.0, 15.0) RETRYABLE_HTTP_STATUSES = frozenset({404, 408, 409, 429, 500, 502, 503, 504}) +# Static price lists in USD per million tokens for providers that return token +# usage but not a per-request cost. DeepSeek's official endpoint is the current +# case: it reports ``input_tokens``, ``output_tokens``, and cache token counts, +# but no ``cost`` field and no generation id, so the existing usage and +# generation reconciliation paths cannot resolve a cost. Update these numbers +# when the provider changes its pricing. DeepSeek's cache-write price is zero. +_TOKEN_PRICES: dict[str, dict[str, Decimal]] = { + "deepseek-flash": { + "input": Decimal("0.3"), + "output": Decimal("1.2"), + "cache_read": Decimal("0.006"), + "cache_write": Decimal("0"), + }, +} +_TOKENS_PER_PRICE_UNIT = Decimal(1_000_000) + + +def estimated_cost(model: object, usage: JsonObject | None) -> JsonObject | None: + """Estimate a USD cost from token usage for a priced, non-reporting model. + + Returns ``None`` when the model has no local price list or the usage does + not carry the integer token counts the estimate needs. The result is marked + ``kind: estimated`` so it stays distinguishable from a provider-reported or + reconciled cost. Call this only after :func:`usage_cost` fails, so an + explicit provider cost always wins. + """ + if not isinstance(model, str) or not isinstance(usage, dict): + return None + prices = _TOKEN_PRICES.get(model) + if prices is None: + return None + if "input_tokens" not in usage or "output_tokens" not in usage: + return None + counts: dict[str, int] = {} + for field in ( + "input_tokens", + "output_tokens", + "cache_read_input_tokens", + "cache_creation_input_tokens", + ): + value = usage.get(field, 0) + if isinstance(value, bool) or not isinstance(value, int) or value < 0: + return None + counts[field] = value + amount = ( + Decimal(counts["input_tokens"]) * prices["input"] + + Decimal(counts["cache_read_input_tokens"]) * prices["cache_read"] + + Decimal(counts["cache_creation_input_tokens"]) * prices["cache_write"] + + Decimal(counts["output_tokens"]) * prices["output"] + ) / _TOKENS_PER_PRICE_UNIT + if not amount.is_finite() or amount < 0: + return None + return { + "status": "resolved", + "amount": str(amount), + "currency": "USD", + "source": "token_estimate.deepseek", + "kind": "estimated", + } + def generation_url(base_url: object) -> str | None: """Build the provider-local generation endpoint from an SDK base URL.""" diff --git a/tests/test_cost.py b/tests/test_cost.py index d827378..9d7fb22 100644 --- a/tests/test_cost.py +++ b/tests/test_cost.py @@ -4,6 +4,7 @@ import pytest from nanopycodeagent.cost import ( + estimated_cost, generation_url, pending_cost, resolve_generation_cost, @@ -39,6 +40,52 @@ def test_usage_cost_preserves_provider_reported_decimal(): } +def test_estimated_cost_prices_deepseek_flash_token_usage(): + assert estimated_cost( + "deepseek-flash", + { + "input_tokens": 5719, + "output_tokens": 2203, + "cache_read_input_tokens": 25600, + "cache_creation_input_tokens": 0, + }, + ) == { + "status": "resolved", + "amount": "0.0045129", + "currency": "USD", + "source": "token_estimate.deepseek", + "kind": "estimated", + } + + +def test_estimated_cost_defaults_absent_cache_counts_to_zero(): + # A minimal DeepSeek-style usage object still prices without cache fields. + assert estimated_cost( + "deepseek-flash", {"input_tokens": 1000, "output_tokens": 1000} + ) == { + "status": "resolved", + "amount": "0.0015", + "currency": "USD", + "source": "token_estimate.deepseek", + "kind": "estimated", + } + + +@pytest.mark.parametrize( + ("model", "usage"), + [ + ("unknown-model", {"input_tokens": 1, "output_tokens": 1}), + ("deepseek-flash", None), + ("deepseek-flash", {}), + ("deepseek-flash", {"input_tokens": 1}), + ("deepseek-flash", {"input_tokens": 1, "output_tokens": True}), + ("deepseek-flash", {"input_tokens": -1, "output_tokens": 1}), + ], +) +def test_estimated_cost_is_unknown_without_a_price_or_valid_counts(model, usage): + assert estimated_cost(model, usage) is None + + def test_generation_resolution_retries_until_cost_is_available(): responses = iter( [ From f04913d9d46c16dff892d2a23adc2a694690fd8e Mon Sep 17 00:00:00 2001 From: minixalpha Date: Thu, 1 Oct 2026 20:29:02 +0800 Subject: [PATCH 2/3] fix(harbor): retry a transient 404 during agent setup Pinned task images occasionally resolve a package index that still points at removed .deb versions, so apt-get install fails with a 404 during agent setup before any model call. Retry the system-dependency transaction a bounded number of times; non-404 failures and the final attempt still propagate. --- .../harbor/src/harbor_adapter/adapter.py | 38 +++++++++- benchmarks/harbor/tests/test_adapter.py | 74 +++++++++++++++++++ 2 files changed, 108 insertions(+), 4 deletions(-) diff --git a/benchmarks/harbor/src/harbor_adapter/adapter.py b/benchmarks/harbor/src/harbor_adapter/adapter.py index 621eebb..94141ce 100644 --- a/benchmarks/harbor/src/harbor_adapter/adapter.py +++ b/benchmarks/harbor/src/harbor_adapter/adapter.py @@ -1,5 +1,6 @@ """Harbor adapter for running nanoPyCodeAgent in benchmark containers.""" +import asyncio import re import shlex import uuid @@ -8,6 +9,7 @@ from harbor.agents.installed.base import ( BaseInstalledAgent, CliFlag, + NonZeroAgentExitCodeError, with_prompt_template, ) from harbor.agents.model_connection import ModelConnectionSpec @@ -16,6 +18,11 @@ from harbor.models.trajectories.trajectory import Trajectory _DEFAULT_MAX_TURNS = 50 +_SYSTEM_DEPENDENCIES = ("curl", "bash", "git", "python3", "ca_certificates") +# Pinned task images occasionally resolve a package index that still points at +# removed `.deb` versions, producing a transient 404 during `apt-get install`. +# Retry the whole transaction a couple of times before failing setup. +_DEPENDENCY_RETRY_DELAYS = (2.0, 4.0) _PACKAGE_NAME = "nanoPyCodeAgent" _REPOSITORY_URL = "https://github.com/minixalpha/nanoPyCodeAgent.git" _UV_VERSION = "0.9.11" @@ -86,10 +93,7 @@ def _install_target(self) -> str: @override async def install(self, environment: BaseEnvironment) -> None: - await self.ensure_system_dependencies( - environment, - ("curl", "bash", "git", "python3", "ca_certificates"), - ) + await self._ensure_system_dependencies(environment) install_target = shlex.quote(self._install_target()) await self.exec_as_agent( environment, @@ -104,6 +108,32 @@ async def install(self, environment: BaseEnvironment) -> None: ), ) + async def _ensure_system_dependencies( + self, environment: BaseEnvironment + ) -> None: + """Install base dependencies, retrying a transient package-index 404. + + The retry only covers the system-package step, so it cannot replay any + model work; a non-404 failure or the final attempt propagates. + """ + total_attempts = len(_DEPENDENCY_RETRY_DELAYS) + 1 + for attempt, delay in enumerate((*_DEPENDENCY_RETRY_DELAYS, None)): + try: + await self.ensure_system_dependencies( + environment, _SYSTEM_DEPENDENCIES + ) + return + except NonZeroAgentExitCodeError as error: + if delay is None or "404" not in str(error): + raise + self.logger.warning( + "Retrying system dependency installation after HTTP 404 " + "(attempt %s/%s); no model call has started.", + attempt + 1, + total_attempts, + ) + await asyncio.sleep(delay) + def _runtime_env(self) -> dict[str, str]: model_connection = self.model_connection env: dict[str, str] = {} diff --git a/benchmarks/harbor/tests/test_adapter.py b/benchmarks/harbor/tests/test_adapter.py index a4c5a4a..3c59e18 100644 --- a/benchmarks/harbor/tests/test_adapter.py +++ b/benchmarks/harbor/tests/test_adapter.py @@ -6,6 +6,7 @@ from types import SimpleNamespace import pytest +from harbor.agents.installed.base import NonZeroAgentExitCodeError from harbor.models.agent.context import AgentContext from harbor_adapter import NanoPyCodeAgent @@ -103,6 +104,79 @@ def test_install_source_rejects_a_revision_coerced_to_a_number(tmp_path): make_adapter(tmp_path, git_ref=float("inf")) +def test_setup_retries_a_transient_404_and_then_succeeds(tmp_path, monkeypatch): + import harbor_adapter.adapter as adapter_module + + adapter = make_adapter(tmp_path) + attempts = [] + sleeps = [] + + async def fake_sleep(delay): + sleeps.append(delay) + + async def flaky(environment, dependencies): + attempts.append(dependencies) + if len(attempts) < 3: + raise NonZeroAgentExitCodeError( + "Command failed (exit 100): apt-get install -y curl\n" + "E: Failed to fetch http://deb.debian.org/.../curl.deb 404 Not Found" + ) + + monkeypatch.setattr(adapter, "ensure_system_dependencies", flaky) + monkeypatch.setattr(adapter_module.asyncio, "sleep", fake_sleep) + + asyncio.run(adapter._ensure_system_dependencies(RecordingEnvironment())) + + assert len(attempts) == 3 + assert sleeps == [2.0, 4.0] + + +def test_setup_does_not_retry_a_non_404_failure(tmp_path, monkeypatch): + import harbor_adapter.adapter as adapter_module + + adapter = make_adapter(tmp_path) + attempts = [] + sleeps = [] + + async def fake_sleep(delay): + sleeps.append(delay) + + async def failing(environment, dependencies): + attempts.append(dependencies) + raise NonZeroAgentExitCodeError("Command failed (exit 1): false") + + monkeypatch.setattr(adapter, "ensure_system_dependencies", failing) + monkeypatch.setattr(adapter_module.asyncio, "sleep", fake_sleep) + + with pytest.raises(NonZeroAgentExitCodeError): + asyncio.run(adapter._ensure_system_dependencies(RecordingEnvironment())) + + assert len(attempts) == 1 + assert sleeps == [] + + +def test_setup_gives_up_after_the_last_retry(tmp_path, monkeypatch): + import harbor_adapter.adapter as adapter_module + + adapter = make_adapter(tmp_path) + attempts = [] + + async def fake_sleep(delay): + pass + + async def always_404(environment, dependencies): + attempts.append(dependencies) + raise NonZeroAgentExitCodeError("E: 404 Not Found") + + monkeypatch.setattr(adapter, "ensure_system_dependencies", always_404) + monkeypatch.setattr(adapter_module.asyncio, "sleep", fake_sleep) + + with pytest.raises(NonZeroAgentExitCodeError): + asyncio.run(adapter._ensure_system_dependencies(RecordingEnvironment())) + + assert len(attempts) == 3 + + def test_run_pipes_the_instruction_and_forwards_anthropic_configuration(tmp_path): instruction = 'fix "quoted" input; echo $TOKEN\nthen run the tests' adapter = make_adapter( From ff6f6e48c3ac7f507e95db8df56e292f0ec57da1 Mon Sep 17 00:00:00 2001 From: minixalpha Date: Thu, 1 Oct 2026 20:29:02 +0800 Subject: [PATCH 3/3] docs(benchmarks): record the DeepSeek official pilot20 Add the structured result and report for running the 20-task pilot set on the DeepSeek official endpoint with deepseek-flash (16/18 scored, two infrastructure exceptions, token-estimated cost), plus the changelog entries for the cost estimator and the setup retry. --- .../reports/deepseek-official-20260930.md | 86 ++++ ...21-deepseek-official-pilot20-20260930.json | 387 ++++++++++++++++++ docs/changelogs/0.8.x.md | 11 + 3 files changed, 484 insertions(+) create mode 100644 benchmarks/harbor/reports/deepseek-official-20260930.md create mode 100644 benchmarks/harbor/results/tb21-deepseek-official-pilot20-20260930.json diff --git a/benchmarks/harbor/reports/deepseek-official-20260930.md b/benchmarks/harbor/reports/deepseek-official-20260930.md new file mode 100644 index 0000000..4b2af9c --- /dev/null +++ b/benchmarks/harbor/reports/deepseek-official-20260930.md @@ -0,0 +1,86 @@ +# DeepSeek official endpoint pilot20 + +Switched the 20-task pilot set from OpenRouter to DeepSeek's official +Anthropic-compatible endpoint and re-ran it with `deepseek-flash`. The run +scored **16/18 scored trials** (Harbor mean 0.80); two trials were lost to +infrastructure failures rather than model failures. + +The complete configuration, per-trial usage, token-estimated cost, and balance +snapshots are in the +[public results](../results/tb21-deepseek-official-pilot20-20260930.json). Raw +evidence remains in the recorded Git-ignored `jobs/` directories. + +## Configuration + +| Setting | Value | +| --- | --- | +| Job | `tb21-deepseek-official-pilot20-20260930` | +| Agent | `e75cacb6` (`0.8.1.dev45`), locally built and verified wheel | +| Endpoint | `https://api.deepseek.com/anthropic` (Anthropic Messages) | +| Model | `deepseek-flash` | +| Tasks | the pinned 20-task pilot set | +| Budget | max_turns 100, max_tokens 65536, per-task agent budget native−180 s | +| Concurrency / attempts / retries | 2 / 1 / 0 | +| Provider pin / HTTP recorder | none — the provider is the only change | + +## Result + +| Bucket | Count | +| --- | ---: | +| Planned trials | 20 | +| Scored trials | 18 | +| Passed | 16 | +| Reward 0 | 2 | +| Infrastructure exceptions | 2 | +| Pass rate over scored | 88.9% | + +- Reward 0: `torch-tensor-parallelism`, `pytorch-model-recovery`. +- `qemu-alpine-ssh`: agent setup failed fetching pinned Debian packages with + HTTP 404; no model call happened. +- `torch-pipeline-parallelism`: the agent completed, but the official verifier + exceeded its native 900 s limit. + +Token usage: 30,933,108 prompt (30,586,240 cached, 346,868 uncached), 891,020 +completion. + +## Cost without a provider cost API + +DeepSeek's official API reports token usage but no per-request cost, no +generation id, and no usage/billing endpoint — `usage.cost` is absent, +`x-generation-id` is absent, and `/v1/generation`, `/user/usage`, +`/billing/usage`, and `/pricing` all return 404. The previous OpenRouter +reconciliation path therefore cannot resolve a DeepSeek cost. + +This branch adds a **token-based estimator** (see Implementation) that prices +the returned tokens with a static DeepSeek price list (USD per million: +input 0.3, cache read 0.006, cache write 0, output 1.2). It marks the result +`kind: estimated`, so it stays distinguishable from provider-reported cost. + +| Component | Tokens | Rate | Cost (USD) | +| --- | ---: | ---: | ---: | +| Uncached input | 346,868 | 0.3 | 0.104060 | +| Cache read | 30,586,240 | 0.006 | 0.183517 | +| Output | 891,020 | 1.2 | 1.069224 | +| **Total** | | | **1.356802** | + +An account balance delta (`/user/balance`, ¥106.63 → ¥102.01) is recorded as an +upper bound only: the interactive assistant session used the same account +during the run, so it is not an isolated run cost, and it is denominated in CNY. + +## Implementation + +- `cost.py`: add `estimated_cost(model, usage)` and a static DeepSeek price + table; `agent.py` uses it only after `usage_cost` fails, so an explicit + provider cost always wins. +- `harbor_adapter`: retry the system-dependency `apt-get` transaction once or + twice on a transient HTTP 404 before failing setup. In the follow-up qemu + re-run the retry fired on attempts 1/3 and 2/3, but the pinned + `bullseye-security` index references `.deb` versions that are gone, so the 404 + is deterministic and a retry alone cannot fix that task. The historical fix + preloads a verified `.deb` cache from the original APT index. + +## Comparison limits + +The OpenRouter pilot20 runs used provider pinning, a different snapshot date, +and different instrumentation, and this run has two infrastructure exceptions. +This is a provider-switch record, not a controlled causal comparison. diff --git a/benchmarks/harbor/results/tb21-deepseek-official-pilot20-20260930.json b/benchmarks/harbor/results/tb21-deepseek-official-pilot20-20260930.json new file mode 100644 index 0000000..ebd66c2 --- /dev/null +++ b/benchmarks/harbor/results/tb21-deepseek-official-pilot20-20260930.json @@ -0,0 +1,387 @@ +{ + "schema_version": 1, + "generated_at": "2026-10-01T12:28:08.216679+00:00", + "job_name": "tb21-deepseek-official-pilot20-20260930", + "run_date": "2026-09-30", + "scope": "Single-attempt provider switch of the 20-task pilot set from OpenRouter to the DeepSeek official Anthropic-compatible endpoint; not a full Terminal-Bench 2.1 score or a controlled causal comparison.", + "experiment": { + "agent_ref": "e75cacb6a780314471ff12655ecb6fe414109ee4", + "wheel_sha256": "2be853f2dcbaef286295aacfbc2c074455d41b8531d7d585fdc727eeecaa9d73", + "harbor_version": "0.21.0", + "dataset": "terminal-bench/terminal-bench-2-1", + "dataset_ref": "sha256:7d7bdc1cbedad549fc1140404bd4dc45e5fd0ea7c4186773687d177ad3a0699a", + "model": "deepseek/deepseek-flash", + "endpoint": "https://api.deepseek.com/anthropic", + "api": "Anthropic Messages", + "model_settings_source": "~/.nanoPyCodeAgent/settings.json", + "limits": { + "n_tasks": 20, + "n_attempts": 1, + "n_concurrent": 2, + "max_turns": 100, + "max_tokens_per_response": 65536, + "time_budget": "native task timeout - 180s, computed per task", + "harbor_max_retries": 0, + "harbor_timeout_overrides": false, + "openrouter_provider_pin": null, + "http_recorder": null, + "temperature": "not explicitly set", + "reasoning_effort": "not explicitly set" + } + }, + "scores": { + "planned_trials": 20, + "scored_trials": 18, + "passed_trials": 16, + "failed_reward_trials": 2, + "unscored_trials": 2, + "exception_counts": { + "VerifierTimeoutError": 1, + "NonZeroAgentExitCodeError": 1 + }, + "harbor_eval_mean": 0.8, + "pass_rate_over_planned": 0.8, + "pass_rate_over_scored": 0.8888888888888888 + }, + "usage": { + "total_prompt_tokens": 30933108, + "total_cached_tokens": 30586240, + "total_cache_write_tokens": 0, + "total_uncached_input_tokens": 346868, + "total_completion_tokens": 891020, + "complete": true + }, + "cost": { + "currency": "USD", + "method": "token_estimate.deepseek", + "source": "static DeepSeek official price table", + "prices_per_million_tokens": { + "input": "0.3", + "output": "1.2", + "cache_read": "0.006", + "cache_write": "0" + }, + "token_estimated_total_usd": "1.35680184", + "reported_by_provider": false, + "note": "DeepSeek's official API returns token usage but no per-request cost, no generation id, and no usage/billing endpoint. This total is computed from token usage, not reported by the provider.", + "account_balance_snapshots_cny": [ + { + "label": "before", + "at": "2026-09-30T14:05:18.075037+00:00", + "is_available": true, + "balance_infos": [ + { + "currency": "CNY", + "total_balance": "106.63", + "granted_balance": "0.00", + "topped_up_balance": "106.63" + } + ] + }, + { + "label": "after", + "at": "2026-09-30T16:14:44.312896+00:00", + "is_available": true, + "balance_infos": [ + { + "currency": "CNY", + "total_balance": "102.01", + "granted_balance": "0.00", + "topped_up_balance": "102.01" + } + ] + } + ], + "account_balance_delta_cny": "4.62", + "account_balance_caveat": "The interactive assistant session used the same DeepSeek account during the run, so the balance delta is an upper bound, not an isolated run cost." + }, + "trials": [ + { + "task": "terminal-bench/caffe-cifar-10", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 3600, + "usage": { + "prompt_tokens": 932801, + "cache_read_tokens": 914048, + "cache_write_tokens": 0, + "uncached_input_tokens": 18753, + "completion_tokens": 25539 + }, + "estimated_cost_usd": "0.041756988" + }, + { + "task": "terminal-bench/circuit-fibsqrt", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 3600, + "usage": { + "prompt_tokens": 972406, + "cache_read_tokens": 963328, + "cache_write_tokens": 0, + "uncached_input_tokens": 9078, + "completion_tokens": 64174 + }, + "estimated_cost_usd": "0.085512168" + }, + { + "task": "terminal-bench/dna-assembly", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 1800, + "usage": { + "prompt_tokens": 2356256, + "cache_read_tokens": 2331008, + "cache_write_tokens": 0, + "uncached_input_tokens": 25248, + "completion_tokens": 88429 + }, + "estimated_cost_usd": "0.127675248" + }, + { + "task": "terminal-bench/kv-store-grpc", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 32653, + "cache_read_tokens": 29568, + "cache_write_tokens": 0, + "uncached_input_tokens": 3085, + "completion_tokens": 2825 + }, + "estimated_cost_usd": "0.004492908" + }, + { + "task": "terminal-bench/llm-inference-batching-scheduler", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 1800, + "usage": { + "prompt_tokens": 1079376, + "cache_read_tokens": 1061248, + "cache_write_tokens": 0, + "uncached_input_tokens": 18128, + "completion_tokens": 52222 + }, + "estimated_cost_usd": "0.074472288" + }, + { + "task": "terminal-bench/log-summary-date-ranges", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 27223, + "cache_read_tokens": 22656, + "cache_write_tokens": 0, + "uncached_input_tokens": 4567, + "completion_tokens": 1777 + }, + "estimated_cost_usd": "0.003638436" + }, + { + "task": "terminal-bench/merge-diff-arc-agi-task", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 274857, + "cache_read_tokens": 260224, + "cache_write_tokens": 0, + "uncached_input_tokens": 14633, + "completion_tokens": 16171 + }, + "estimated_cost_usd": "0.025356444" + }, + { + "task": "terminal-bench/model-extraction-relu-logits", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 538556, + "cache_read_tokens": 526592, + "cache_write_tokens": 0, + "uncached_input_tokens": 11964, + "completion_tokens": 29094 + }, + "estimated_cost_usd": "0.041661552" + }, + { + "task": "terminal-bench/mteb-leaderboard", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 3600, + "usage": { + "prompt_tokens": 1719176, + "cache_read_tokens": 1680128, + "cache_write_tokens": 0, + "uncached_input_tokens": 39048, + "completion_tokens": 29412 + }, + "estimated_cost_usd": "0.057089568" + }, + { + "task": "terminal-bench/openssl-selfsigned-cert", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 40143, + "cache_read_tokens": 37120, + "cache_write_tokens": 0, + "uncached_input_tokens": 3023, + "completion_tokens": 3902 + }, + "estimated_cost_usd": "0.00581202" + }, + { + "task": "terminal-bench/path-tracing", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 1800, + "usage": { + "prompt_tokens": 1140749, + "cache_read_tokens": 1084672, + "cache_write_tokens": 0, + "uncached_input_tokens": 56077, + "completion_tokens": 35038 + }, + "estimated_cost_usd": "0.065376732" + }, + { + "task": "terminal-bench/pypi-server", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 110863, + "cache_read_tokens": 105216, + "cache_write_tokens": 0, + "uncached_input_tokens": 5647, + "completion_tokens": 5713 + }, + "estimated_cost_usd": "0.009180996" + }, + { + "task": "terminal-bench/pytorch-model-recovery", + "reward": 0.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 1481698, + "cache_read_tokens": 1460352, + "cache_write_tokens": 0, + "uncached_input_tokens": 21346, + "completion_tokens": 44263 + }, + "estimated_cost_usd": "0.068281512" + }, + { + "task": "terminal-bench/qemu-alpine-ssh", + "reward": null, + "exception_type": "NonZeroAgentExitCodeError", + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 0, + "cache_read_tokens": 0, + "cache_write_tokens": 0, + "uncached_input_tokens": 0, + "completion_tokens": 0 + }, + "estimated_cost_usd": "0.000" + }, + { + "task": "terminal-bench/regex-chess", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 3600, + "usage": { + "prompt_tokens": 6867208, + "cache_read_tokens": 6828288, + "cache_write_tokens": 0, + "uncached_input_tokens": 38920, + "completion_tokens": 122473 + }, + "estimated_cost_usd": "0.199613328" + }, + { + "task": "terminal-bench/regex-log", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 435322, + "cache_read_tokens": 429824, + "cache_write_tokens": 0, + "uncached_input_tokens": 5498, + "completion_tokens": 62276 + }, + "estimated_cost_usd": "0.078959544" + }, + { + "task": "terminal-bench/schemelike-metacircular-eval", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 2400, + "usage": { + "prompt_tokens": 11204578, + "cache_read_tokens": 11154688, + "cache_write_tokens": 0, + "uncached_input_tokens": 49890, + "completion_tokens": 192178 + }, + "estimated_cost_usd": "0.312508728" + }, + { + "task": "terminal-bench/torch-pipeline-parallelism", + "reward": null, + "exception_type": "VerifierTimeoutError", + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 1334231, + "cache_read_tokens": 1318656, + "cache_write_tokens": 0, + "uncached_input_tokens": 15575, + "completion_tokens": 62393 + }, + "estimated_cost_usd": "0.087456036" + }, + { + "task": "terminal-bench/torch-tensor-parallelism", + "reward": 0.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 168404, + "cache_read_tokens": 164992, + "cache_write_tokens": 0, + "uncached_input_tokens": 3412, + "completion_tokens": 18918 + }, + "estimated_cost_usd": "0.024715152" + }, + { + "task": "terminal-bench/write-compressor", + "reward": 1.0, + "exception_type": null, + "native_agent_timeout_seconds": 900, + "usage": { + "prompt_tokens": 216608, + "cache_read_tokens": 213632, + "cache_write_tokens": 0, + "uncached_input_tokens": 2976, + "completion_tokens": 34223 + }, + "estimated_cost_usd": "0.043242192" + } + ], + "qemu_alpine_ssh_retry": { + "job": "tb21-deepseek-official-qemu-retry-20260930", + "exception_type": "NonZeroAgentExitCodeError", + "retry": "The tracked 404 retry fired on attempts 1/3 and 2/3 and still failed.", + "finding": "The pinned Debian bullseye-security index references .deb versions that are no longer in the pool; the 404 is deterministic, so a retry alone cannot fix this task. The historical fix preloads a verified .deb cache from the original APT index." + }, + "comparison_limits": "The OpenRouter pilot20 runs used provider pinning, a different snapshot date, and different instrumentation. Two exceptions here are infrastructure failures (apt 404 and a verifier timeout), not model failures." +} diff --git a/docs/changelogs/0.8.x.md b/docs/changelogs/0.8.x.md index 8107711..125ecaf 100644 --- a/docs/changelogs/0.8.x.md +++ b/docs/changelogs/0.8.x.md @@ -5,6 +5,13 @@ All notable changes in the **0.8.x** release series are documented here. ## [Unreleased] ### Added +- Estimate model-call costs from token usage when a provider reports tokens + but no per-request cost, starting with the DeepSeek official + Anthropic-compatible endpoint. The estimate uses a static DeepSeek price + list, is marked `kind: estimated` so it stays distinct from a + provider-reported or reconciled cost, and is only used after a real + provider-reported cost is absent. DeepSeek exposes no cost or generation + endpoint, so the existing OpenRouter reconciliation cannot resolve it. - Configure per-reply generation budgets through `--max-tokens` or `ANTHROPIC_MAX_TOKENS`, including the user settings file and Harbor adapter. Record the effective limit in startup output, Event Journals, and ATIF @@ -25,6 +32,10 @@ All notable changes in the **0.8.x** release series are documented here. supported. ### Fixed +- Retry the Harbor adapter's system-dependency installation a bounded number + of times when the package index returns a transient HTTP 404, instead of + failing agent setup immediately. Non-404 failures and the final attempt still + propagate. - Terminate a timed-out or interrupted bash command's process group on POSIX and other members of its session on Linux, including GNU `timeout` children in separate groups. Close output pipes without draining detached children