diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index c23f0d7..a943ab5 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -38,12 +38,12 @@ { "name": "crypto-data", "source": "./skills/crypto-data", - "description": "Use for any crypto data question — token prices, FX, commodities, stocks, OHLC history, DEX pairs, DeFi TVL and yields, on-chain SQL, wallet labels and net worth, social mindshare. Routes across five overlapping tools and says which one to use and which are free." + "description": "Use for any crypto data question — token prices, FX, commodities, stock ticker catalog, OHLC history, DEX pairs, DeFi TVL and yields, on-chain SQL, wallet labels and net worth, social mindshare. Routes across five overlapping tools and says which one to use and which are free." }, { "name": "surf", "source": "./skills/surf", - "description": "Use when the user wants deep crypto data — on-chain SQL, CEX order books, wallet labels and net worth, social mindshare, news and unified search across exchange, on-chain, wallet, social and prediction endpoints." + "description": "Surf data endpoints were retired by the gateway on 2026-09-06 — this skill redirects each former Surf question to the tool that still serves it (blockrun_price, blockrun_defi, blockrun_markets, blockrun_dex, blockrun_rpc) and names what has no replacement yet." }, { "name": "rpc", diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c5e7e77..722625d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -14,14 +14,25 @@ concurrency: cancel-in-progress: true jobs: - test: + # package.json says `engines.node >=20.19` and the README carries the badge, + # but until this matrix existed every check ran on Node 22 only. A dependency + # reaching for a 22-only API (Promise.withResolvers, Set.prototype.union) would + # have shipped green and broken on the oldest Node the package claims to + # support. 20.19 is the floor: mock.module (which `npm test` needs) landed in + # 20.18, and vite 8 / rolldown (the pretest MCP Apps build) require ^20.19. + test-node: + name: test (node ${{ matrix.node }}) runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + node: ["20.19", "22"] steps: - uses: actions/checkout@v4 - uses: actions/setup-node@v4 with: - node-version: "22" + node-version: ${{ matrix.node }} cache: npm - run: npm ci @@ -58,3 +69,17 @@ jobs: # a blockrun.ai deploy in progress cannot fail this repo's CI. - name: Brand numbers run: node scripts/sync-brand-numbers.mjs --check + + # Branch protection on main requires a status check named exactly `test`. + # A matrix job reports one check per leg ("test (node 20.19)", "test (node + # 22)") and never one called `test`, so without this gate every PR would sit + # unmergeable waiting for a check that can no longer arrive. `if: always()` + # makes it run even when a leg fails, so the failure is reported as a red + # `test` rather than a check that never completes. + test: + needs: test-node + if: always() + runs-on: ubuntu-latest + steps: + - name: All Node versions passed + run: test "${{ needs.test-node.result }}" = "success" diff --git a/CHANGELOG.md b/CHANGELOG.md index e0ab01b..9d57e4c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,155 @@ All notable changes to BlockRun MCP will be documented in this file. +## 0.49.0 + +**The error says whether money moved.** Issue #132 reported `blockrun_markets` +and `blockrun_price` failing with `API error after payment: 502 / Request failed` +while the wallet balance never changed. The balance was right and the words were +wrong, and the words were wrong on three layers at once. + +The gateway had said exactly what happened — `Predexon 500: An unexpected error +occurred (payment NOT charged)` — but `@blockrun/llm` kept only the top-level +`error` string when it sanitized the body, so the cause and the settlement status +never reached this server. That is fixed in `@blockrun/llm` 3.15.1 (PR #39), which +carries the gateway's message through as `detail`; this release depends on it and +reads the new field, because our own extractor only ever looked at `message` and +`hint` and would have dropped it a second time. The formatter's "not charged" +branch, which already existed, finally receives the string it was written for. + +**Sports is degraded upstream, and the tool now says so.** All four Predexon +`sports/*` routes have returned an upstream 500 on every call since 2026-08-04. +The gateway marks them degraded, withdrew them from discovery, and releases the +payment nonce on upstream failure, so a call costs nothing and returns nothing. +This server still advertised them as live. The description, the prediction-markets +skill and the crypto-data and surf skills now say the routes are degraded and +point at `markets/search` with `{ q: "NBA" }` or `polymarket/events` with +`{ search: "NBA" }` instead; a `sports/*` 5xx renders the outage, the date, and +whether money moved rather than "temporary API issue, try again". The routes +stay callable, because the gateway is the authority on whether Predexon has +recovered. + +An earlier draft of this release steered users to the canonical `markets` route +with a `league` filter. That route, `outcomes/:predexon_id` and +`matching-markets` were removed upstream on 2026-08-04 — they answer 404 +"Unknown Predexon endpoint" before payment — and no live `/v1/pm` route accepts +`league` at all, so the headline remedy would have failed on first use across +eight surfaces. Every one of them now names the two routes that quote a 402 +today (probed unauthenticated, both gateways), and the dead routes are gone from +the tool description and the prediction-markets skill. + +**"Nothing was charged" now needs the gateway's word for it.** The sports +formatter asserted no charge from a 5xx status range alone. Only the gateway's +upstream-failure branch releases the payment nonce, and only that branch writes +"(payment NOT charged)" into the body; its catch-all 500 deliberately does not +release, because settlement ran in the same try, and a 504 after settlement +carries no body at all. So the sentence is now gated on that evidence — "not +charged", "no charge was made", "no payment was made", or the 502's "Upstream +provider error" — and any other labelled 5xx on a sports path keeps the outage +explanation and the steer but says to check `blockrun_wallet action:"report"` +instead, the same hedge a post-payment 501 already gets. Today every wallet-rail +sports failure comes from the release branch, so the wording changes for nobody; +it would have been wrong on exactly the day Predexon recovers and a settle-side +error follows. + +**Surf is gone, and so is `blockrun_surf`.** The gateway has answered every +`/v1/surf/*` path with HTTP 410 `endpoint_retired` since 2026-09-06 (`retired_on` +in the body; `sol.blockrun.ai` 404s; `/api/openapi` lists no Surf route). No 402 +is ever issued, so no payment could have been made — but the tool still reserved +$0.0095 and, with `BLOCKRUN_CONFIRM_SPEND=on`, asked the user to approve a charge +for a route that cannot succeed, while the SDK reduced the gateway's dated, +reasoned notice to `API error: 410 — API request failed`. This release first made +the tool answer with the retirement before any reservation, and then removed it: +a tool that can only return an error is not worth the schema every agent carries +on every turn, and the published brand artifact has said 19 tools since the +delisting. The server now ships **19 tools**, the `trading` profile 8 and +`research` 5. Former Surf questions route to `blockrun_price`, `blockrun_defi`, +`blockrun_markets`, `blockrun_dex` and `blockrun_rpc`; the `surf` skill is now a +map from each former endpoint to its replacement, and says plainly what has none +yet (on-chain SQL, cross-chain wallet labels, CEX books, social mindshare). +README, the crypto-data, gentech, rpc, blockrun and debug skills stop selling 83 +endpoints at $0.0095. A config that still names the tool gets an unknown-tool +error from its MCP client; nothing can be charged either way. + +**Prices say base plus fee, once, everywhere.** `blockrun_markets`, `blockrun_exa` +and `blockrun_defi` hand-typed a "charged" figure in their descriptions that was +the reserve ($0.002 fee), not the charge; the live `payment-required` header +decodes to base + $0.001 on Base and base alone on Solana, and README disagreed +with itself about which it was quoting. `blockrun_rpc` went the other way and +quoted the bare $0.002 base while its own reserve and confirm dialog show $0.004. +Every description and README row now states the base and says the gateway adds +its flat network fee ($0.001 today; $0.002 reserved; the 402 header carries the +exact amount). The reserve constants are untouched — they are the conservative +gate and were right. + +**The registry manifest told Solana users to put a bs58 key in the EVM slot.** +`server.template.json`, stamped into the MCP registry's `server.json` at publish, +described `BLOCKRUN_WALLET_KEY` as "hex for Base, bs58 for Solana". The code reads +Solana keys only from `SOLANA_WALLET_KEY`; a bs58 key in `BLOCKRUN_WALLET_KEY` +selects Base and dies in viem's hex parser on the first paid call. The manifest +now describes `BLOCKRUN_WALLET_KEY` as the 0x-hex EVM key and Polymarket signer, +and lists `SOLANA_WALLET_KEY` and `BLOCKRUN_API_KEY` alongside it. + +**Equity quotes are not served, and the tool no longer sells them.** Since +2026-09-05 the gateway answers every `stocks/{market}/price` and `history` call +(and the `usstock` alias) with a pre-payment 501: "We do not currently serve +equity prices." This server's description, README and two skills still promised +paid stock quotes, and on the default Solana chain the Base-only guard fired first +and told the user to switch chains to pay for a route that cannot succeed. Paid +stock calls now return the gateway's own answer before the wallet is consulted: +withdrawn on 2026-09-05, nothing charged, the ticker catalog is still free, and +who to contact for equity coverage. `formatError` also stops labelling any 501 a +transient outage; it claims "nothing was charged" only when the 501 arrived +before payment. + +**Pay what you were told, or nothing.** `npm run verify:prices` caught a third +layer while this release was being cut: the Solana gateway is a separate +deployment that can lag Base, and it does not know `azure/sora-2` — it quotes +"Seedance 2.0 Pro video generation (5s)" at $1.135 in Sora's place, 2.7x the +published rate, for a different model. The only check on the gateway's price was +the budget cap, which would have let that through on any wallet holding $2. +`blockrun_video` (both rails) and `blockrun_image` (Solana) now compare the 402 +against the estimate the model was shown and refuse, unsigned, anything more than +1.5x above it; the message names the quoted amount, what the gateway labelled it, +and how to proceed. The Solana helper hands callers the decoded 402 so they can +judge what was quoted, not just how much. The price verifier classifies a Solana +quote for a different product as a gateway bug to report rather than an +estimator gap to paper over. + +### From the audit + +With #132 fixed, the whole server went through a ten-angle audit, each finding argued against by an adversarial verifier before it counted. Thirty-seven survived; every one is fixed below, ordered by what it would have cost. + +**A strict-mode Solana wallet could be destroyed by reading its own status.** `ensureBothWallets` — reached by the default `blockrun_wallet` action and by `action:"chain"` — provisioned the Solana side through the SDK's file-only loader. Under `BLOCKRUN_KEYCHAIN=strict` the `.solana-session` file is retired once the key is in the keychain, so that loader saw an empty slate and minted a new keypair; the next key resolution then mirrored the new key over the funded one with `-U` and deleted the file. The funded key was in neither store. The Solana path now has the same shape as the EVM path (`ensureSolanaWallet`: env, then file, then keychain via a read that keeps "absent" and "failed" apart), refuses to mint when the keychain could not be read, and no longer memoises a miss — a wallet provisioned later is visible without a restart. Two more wallet facts came out of the same reading: a fresh install (where Solana is the default) died in the SDK constructor with "Private key required" on every status, setup, QR and deposit call before it ever reached the one action that creates wallets — `getWalletInfo` now provisions, and the client factory names the remedy; and the status screen read the Solana balance from the client's own key rather than the address it was displaying, so it could print one wallet's address beside another's balance — the balance is now queried by address, and an unreachable RPC reads as "unavailable", not $0. + +**Polymarket's money paths got a round of audit hardening.** A market order is now signed at the worst fill the preview showed — the dry-run walks the live book and prints `worst fill ≤ X` (buy) / `≥ X` (sell), and that X becomes the order's limit, so a book that thins between preview and confirm can only fill less, never worse; estimated sell proceeds are the walked total, not size × best bid. `withdraw` validates `to_address` with a strict checksum before any I/O and labels the destination honestly — "your agent wallet" only when it is, otherwise a loud CUSTOM warning — and a bridge error is reported as the bridge, not as CLOB geoblock advice (redeem got the same scoping). A relayer submit whose response is lost now leaves the signed withdrawal tracked with anti-retry guidance instead of inviting a double-send, and a signer rotation no longer inherits the old vault's `deployed:true`, so setup deploys the new vault instead of telling you to bridge funds into an address with no code. `fund` gains an optional `POLYMARKET_MAX_FUND_USD` per-call cap (unset = unchanged), and the order card refuses to place an edited amount until you re-quote. + +**A job that is already paid for is never abandoned, and never called free.** The account rail bills an async video or music job the moment the gateway accepts it, and until now every failure after that point lost the thread: a dropped poll or a stalled upstream fell out of `apiKeyAsyncPost` as a bare error, `isTimeoutError` matched the text, and the tool said "please try again" — advice that submits and bills a second job — while the local ledger booked nothing, because `finally` released the reservation and no one recorded the charge. Both rails now poll through transient disconnects and proxy statuses inside the existing deadline, as the Solana helper already did; and every give-up after a successful submit is a `BilledJobError` that carries the settled cost and the job id, which `blockrun_video` and `blockrun_music` book against the cap and report with the dashboard link and no retry advice. A submit that never answers says the job *may* have been billed — no charge was observed, so none is asserted, and none is denied. On the Base rail, a poll aborted while carrying the payment header can still settle server-side; the tools now say so and book conservatively instead of promising "no payment was taken", and `blockrun_music` books a completed poll before validating its payload, the fix `blockrun_video` received in 0.39.1. `blockrun_realface` books a settled 2xx before checking for `asset_id`. + +**`blockrun_phone` now refuses paths outside `phone/*` and `voice/*` before reserving budget.** Every other passthrough tool concatenates onto a fixed prefix; phone's prefix was `/v1/` itself, so no traversal was needed — `path:"modal/sandbox/create"` with an H100 body ran at phone's $0.012 unknown reserve, clearing any budget cap and showing the confirm dialog a number 16,000× too small. The route the gateway will serve (decoded, lower-cased, query dropped) is what gets classified, so encoded, cased, and query-suffixed spellings of in-namespace routes still pass and no spelling of an out-of-namespace one does. + +**The chat price table said it held every model priced above the $5/$30 default; two flagships had been sitting above it for weeks.** `openai/gpt-6-astra` and `anthropic/claude-fable-5.1` ($10/$50 on both gateways) now have rows, and `npm run verify:prices` sweeps the live catalogue so the eighth cannot go unnoticed — it also fails when a row reads below the live rate, when a "free" model starts costing, and when an Anthropic row over-books the native ledger, which is how `claude-sonnet-5` drops to its real $2/$10 after a 1.5x over-count. A chat call that settled and then stalled mid-stream now says so on every path, without the "needs funding" advice the routing loop's own note used to earn from the formatter. `thinking.budget_tokens` is reserved only for Claude, as the schema always promised. The native ledger reads the gateway's dashed echoes (`claude-fable-5-1`) as their catalogue key instead of prefix-matching a sibling's rate. "Free" is a set, not a vendor: `cohere/north-mini-code` and `poolside/laguna-xs-2.1` reserve $0. And a 5xx the gateway marks "(payment NOT charged)" finally says, in the tool's voice, that nothing was charged (#132). + +**The tests can no longer spend, and the guards can no longer skip.** `image-cost.test.ts` said the paid client was mocked, and it was — but `blockrun_image` decides its rail from the account key before it ever asks for that client, so on a developer machine set up for account mode the suite left the mocks and posted to the gateway with the real key. The rail is now pinned to a temp `HOME` with no key before the tool loads, and the shared fetch helper is a trap, so an escape fails for the right reason. The confirm-spend guard keyed on `reserveBudget` and skipped any file without one — the one offender it could not see was a tool that pays and never reserves; it now reads the payment surfaces off the imports and holds every one of them to reserve *and* confirm, and `blockrun_image` finally has its own row in the decline table. `blockrun_price`'s equity pre-flight is proved at the handler, not as a string: before the chain guard, the budget gate and the confirm dialog. Two smaller honesty fixes ride along: a `BLOCKRUN_BUDGET_LIMIT` that does not parse (`5,00`, `0`, `5 USD`) now says on stderr that the cap is OFF instead of silently running unlimited, and the image-edit confirm dialog names the real path of every local file about to leave the machine — a symlink is shown as its target. + +**The startup key scanner no longer tells a user who followed the docs to rotate their wallet.** `BLOCKRUN_WALLET_KEY` / `SOLANA_WALLET_KEY` under `mcpServers.*.env` is the documented override — on Claude Code it is the only way to set it — yet every launch printed the "treat this key as compromised" banner. That location now gets a short note (the file is plaintext and synced; prefer `~/.blockrun/.session` or the OS keychain), the banner is reserved for a key somewhere it was never meant to be, a raw key hiding in `args` is finally caught, and the Cursor and Windsurf config files the README documents are scanned too. Around it, four smaller honesty fixes: an unknown `--profile` says so instead of quietly loading all 20 tools (and whitespace/case no longer count as a typo); the update notice stops recommending a `claude mcp add` that refuses an existing name; `blockrun_dex` validates the token address before it goes into the URL path; and `skills install --help` exits 0. CI now runs the full suite on Node 20.19 as well as 22, which `engines` has claimed since the badge went up. + +Also shipping, landed on `main` since 0.48.0: + +- **`blockrun_image` reads the settled cost on the account rail** instead of an + estimate that was high by the transaction fee the rail does not charge, and + stops labelling an exact figure "estimated" (#140). +- **OpenClaw is verified** end-to-end on 2026.8.2 with the published `npx` + package, with install notes on spend confirmation per chat surface; the + `deepseek/deepseek-v4-pro` rate follows the gateway's repricing (#131). +- Brand numbers refreshed from the canonical snapshot (#141). + +Two adversarial passes (a fresh-context Claude subagent and Codex) reviewed the +change; every finding was addressed, including the two that mattered: a +post-payment 501 must not claim nothing was charged, and the sports matcher must +use the same labelled-status rule as `formatError` so an incidental "501 items" +in a 4xx body is not sold as the outage. + ## 0.48.0 **A key can live in a file, not just an environment variable.** Write it to diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 93a8d07..17e3900 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -19,7 +19,7 @@ Smoke-test the built server via the MCP stdio handshake: (printf '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2024-11-05","capabilities":{},"clientInfo":{"name":"test","version":"1"}}}\n{"jsonrpc":"2.0","method":"notifications/initialized"}\n{"jsonrpc":"2.0","id":2,"method":"tools/list"}\n'; sleep 2) | node dist/index.js 2>/dev/null ``` -Should return 20 tools including `blockrun_surf` and any new one you add. +Should return 19 tools including any new one you add. To test locally with Claude Code, point it at your dev build: diff --git a/README.md b/README.md index 26392c7..dc6d491 100644 --- a/README.md +++ b/README.md @@ -6,13 +6,13 @@

Agents can't sign up for accounts. Agents can't enter credit cards.
Agents can only sign transactions.

-BlockRun MCP gives your agent 20 tools — markets, research, web search, images, video, on-chain data, and live Polymarket trading — paid per call.

+BlockRun MCP gives your agent 19 tools — markets, research, web search, images, video, on-chain data, and live Polymarket trading — paid per call.

Two ways to pay, same tools: a self-custody wallet (USDC on Solana or Base, no account needed) — or a BlockRun API key for teams that can't run wallets. Sign up at user.blockrun.ai →

Read the odds and place the bet, from one self-custody wallet.


-20 tools  +19 tools  Agent native  Wallet or API key  Read and trade Polymarket  @@ -44,7 +44,7 @@ claude mcp add blockrun -s user -- npx -y @blockrun/mcp@latest
- Context cost: 12.9K tokens, 6% of a 200K context window, charged every turn whether or not you call a tool. 5.6K with --profile trading, 57% less. + Context cost: 12.7K tokens, 6% of a 200K context window, charged every turn whether or not you call a tool. 5.2K with --profile trading, 59% less.
@@ -52,7 +52,7 @@ claude mcp add blockrun -s user -- npx -y @blockrun/mcp@latest --- -> **BlockRun MCP** is an open-source [Model Context Protocol](https://modelcontextprotocol.io) server that gives Claude — and any MCP-compatible agent — 20 tools for real-time data and real actions: 76 LLMs, image & video generation, prediction-market data, live web/X search, on-chain queries across 40 chains, and **the ability to place real, USDC-settled bets on Polymarket**. +> **BlockRun MCP** is an open-source [Model Context Protocol](https://modelcontextprotocol.io) server that gives Claude — and any MCP-compatible agent — 19 tools for real-time data and real actions: 78 LLMs, image & video generation, prediction-market data, live web/X search, on-chain queries across 40 chains, and **the ability to place real, USDC-settled bets on Polymarket**. You pay per call, and you choose how. **Wallet mode** authenticates with a signature and settles each call in USDC via the [x402](https://x402.org) protocol — no account, no credit card, no subscription, on Solana or Base. **Account mode** authenticates with a BlockRun API key (`brk_live_…`) from [user.blockrun.ai](https://user.blockrun.ai) and bills prepaid credit at exact usage — for teams that can't hand a wallet to an agent. Same 20 tools either way. MIT licensed. @@ -68,7 +68,7 @@ Every other data integration was built for **human developers** — create an ac **Agents can't do any of that.** BlockRun MCP is built for the agent-first world: -- **One wallet, every source** — 20 tools behind a single self-custody wallet. No per-vendor signups. +- **One wallet, every source** — 19 tools behind a single self-custody wallet. No per-vendor signups. - **No API key required** — your wallet signature *is* authentication. (One is available at [user.blockrun.ai](https://user.blockrun.ai) for teams who need an invoice instead of a keypair.) - **No credit cards** — pay per request in USDC via [x402](https://x402.org), fractions of a cent each. - **Starts free** — the free tier (`blockrun_chat mode:"free"`, `blockrun_dex`, crypto `blockrun_price`, `blockrun_models`) costs $0. @@ -85,7 +85,7 @@ Every other data integration was built for **human developers** — create an ac | ------------------- | -------------------------------- | ------------------------- | ----------------------------------------- | | **Setup** | Account + API key *per vendor* | Account/key for 1 vendor | **Wallet auto-created — or one key for everything** | | **Payment** | Credit card, monthly minimums | Credit card / vendor plan | **USDC per-call via x402, or prepaid credit** | -| **Data sources** | One per integration | One vendor | **20 tools — LLMs, media, markets, chain**| +| **Data sources** | One per integration | One vendor | **19 tools — LLMs, media, markets, chain**| | **Place real bets** | Build it yourself | Rare | **Yes — Polymarket CLOB, confirm-gated** | | **Pay-chain** | — | — | **Solana + Base (or no chain at all)** | | **Agent budgets** | Manual | — | **Built-in per-agent delegation** | @@ -124,7 +124,7 @@ After BlockRun, it can. Each query costs fractions of a cent — billed from a l | Best for | Agents, solo devs, anything self-custody | Teams, companies, anyone who can't run a wallet | | Trade on Polymarket | ✅ | ❌ — needs a keypair to sign | -Both modes reach the same 20 tools. You can switch at any time; setting `BLOCKRUN_API_KEY` takes priority over a wallet, and unsetting it hands the wallet back. +Both modes reach the same 19 tools. You can switch at any time; setting `BLOCKRUN_API_KEY` takes priority over a wallet, and unsetting it hands the wallet back. ### 1. Install @@ -157,7 +157,7 @@ claude mcp add blockrun -s user -- npx -y @blockrun/mcp@latest Any other MCP client that can spawn a stdio server works the same way: `command: npx`, `args: ["-y", "@blockrun/mcp@latest"]`. With nvm/Homebrew Node on a JSON-configured client, put the absolute path from `which npx` in `command`. Spend-dialog sources and what "proceeds without asking" means: [`docs/spend-confirmation.md`](docs/spend-confirmation.md). -**OpenClaw:** the published `npx` package was verified end-to-end on 2026.8.2: all 20 tools were projected, free calls worked, and paid x402 calls settled. Add a hard session cap while installing: +**OpenClaw:** the published `npx` package was verified end-to-end on 2026.8.2: all 20 tools were projected (19 since the Surf delisting), free calls worked, and paid x402 calls settled. Add a hard session cap while installing: ```bash openclaw mcp set blockrun '{"command":"npx","args":["-y","@blockrun/mcp@latest"],"env":{"BLOCKRUN_BUDGET_LIMIT":"2"}}' @@ -193,12 +193,14 @@ Expose a trimmed tool set so the client loads fewer schemas into context. Pass ` | Profile | Tools | |---------|-------| -| `full` *(default)* | everything (20 tools) | +| `full` *(default)* | everything (19 tools) | | `media` | `wallet` `models` `image` `video` `realface` `music` `speech` | -| `trading` | `wallet` `price` `dex` `markets` `surf` `defi` `rpc` `polymarket_read` `polymarket` | -| `research` | `wallet` `models` `chat` `search` `exa` `surf` | +| `trading` | `wallet` `price` `dex` `markets` `defi` `rpc` `polymarket_read` `polymarket` | +| `research` | `wallet` `models` `chat` `search` `exa` | | `chat` | `wallet` `models` `chat` | +`blockrun_surf` was removed in 0.49.0: the gateway has answered every Surf (asksurf.ai) path with 410 since 2026-09-06, so the tool could only ever return an error, and its schema cost every agent context on every turn. Those questions go to `blockrun_price`, `blockrun_defi`, `blockrun_markets` and `blockrun_dex`; the `surf` skill maps each former endpoint to its replacement. + ```bash claude mcp add blockrun-trading -s user -- npx -y @blockrun/mcp@latest --profile trading @@ -216,13 +218,13 @@ Package managers have shown install size for decades. Almost no MCP server shows | Profile | Tools | Context | |---------|-------|---------| -| `full` *(default)* | 20 | 12,991 | -| `trading` | 9 | 5,605 | -| `media` | 7 | 5,527 | -| `research` | 6 | 3,075 | -| `chat` | 3 | 1,975 | +| `full` *(default)* | 19 | 12,657 | +| `trading` | 8 | 5,160 | +| `media` | 7 | 5,603 | +| `research` | 5 | 2,635 | +| `chat` | 3 | 1,976 | -Running `--profile trading` instead of the default costs **57% less context** for the same trading +Running `--profile trading` instead of the default costs **59% less context** for the same trading workflow. If you only ever ask about markets, that is the single cheapest change you can make. Measure it yourself — against us, or against any other stdio MCP server: @@ -318,7 +320,7 @@ npx -y @blockrun/mcp@latest skills install --to ~/.codex/skills > **Claude:** According to Polymarket, the market puts a **73% probability** on the Fed holding rates steady, 24% on a 25bp cut, 3% on a hike. 24h volume: $2.1M. The "Hold" contract last traded at $0.73. > -> *(via `blockrun_markets` · cost: $0.0095)* +> *(via `blockrun_markets` · cost: $0.0085 on Base — $0.0075 + the $0.001 network fee)* --- @@ -334,40 +336,41 @@ npx -y @blockrun/mcp@latest skills install --to ~/.codex/skills | Tool | Data source | Cost | |------|-------------|------| -| `blockrun_chat` | 76 LLMs (GPT, Claude, Gemini, DeepSeek, Kimi K3, GLM, NVIDIA free tier, …) with `mode` tier routing | per token | +| `blockrun_chat` | 78 LLMs (GPT, Claude, Gemini, DeepSeek, Kimi K3, GLM, NVIDIA free tier, …) with `mode` tier routing | per token | | `blockrun_image` | Generate: openai/gpt-image-2, gpt-image-1, google/nano-banana(-2/-pro), xai/grok-imagine-image(-pro), zai/cogview-4, bytedance/seedream-5-pro. Edit: img2img, inpaint, fusion. | $0.015–0.15 | | `blockrun_video` | Sora 2 + xAI Grok Imagine Video + ByteDance Seedance 1.5/2.0-mini/2.0-fast/2.0/2.5 (720p + audio; 4K on 2.0, up to 30s on 2.5); RealFace asset → real-person video | $0.053–0.32/sec charged | | `blockrun_realface` | Enroll a real person (phone liveness) or AI character (Virtual Portrait) as a `ta_xxxx` asset for Seedance 2.0 / 2.0-fast / 2.0-mini video (not 2.5) | free; $0.01 to enroll | | `blockrun_music` | MiniMax music generation | per track | | `blockrun_speech` | ElevenLabs TTS (Flash/Turbo/Multilingual/v3, 8 voices) + ByteDance Seed Audio (prompt-directed) + cinematic sound effects; free voice listing | $0.05–0.10/1k chars | -| `blockrun_price` | Pyth-backed realtime + OHLC — crypto / FX / commodity (free), 12 stock markets (paid) | free or $0.001/call | -| `blockrun_markets` | Polymarket (markets, candles, trades, orderbooks, leaderboards, smart-wallet PnL/clusters, UMA oracle), Kalshi, Limitless, Opinion, Predict.Fun, dFlow, Binance Futures, cross-platform search | $0.0095/query | +| `blockrun_price` | Pyth-backed realtime + OHLC — crypto / FX / commodity, plus the ticker catalog for 12 equity markets (equity quotes withdrawn 2026-09-05) | free | +| `blockrun_markets` | Polymarket (markets, candles, trades, orderbooks, leaderboards, smart-wallet PnL/clusters, UMA oracle), Kalshi, Limitless, Opinion, Predict.Fun, dFlow, Binance Futures, cross-platform search | $0.0075 + fee/query | | `blockrun_polymarket_read` | Read-only Polymarket positions/open orders plus executable live order previews, separated for MCP clients that enforce tool safety annotations | free | | `blockrun_polymarket` | **Trade on Polymarket** (CLOB V2): place/cancel real bets, positions, redeem winnings — signed locally, settled in pUSD from a gasless deposit wallet. Confirm-gated, $25/order default cap. [Details ↓](#-polymarket-trading) | free tool; bets are your funds | -| `blockrun_surf` | Surf (asksurf.ai) — 83 endpoints: CEX data, on-chain SQL (13 chains, 80+ tables), 100M+ labeled wallets, Polymarket + Kalshi, social mindshare, news, Surf-1.5 chat with citations | $0.0095/call | -| `blockrun_exa` | Neural web search (Exa) — research, competitors, papers, URL content | $0.01/query | +| `blockrun_exa` | Neural web search (Exa) — research, competitors, papers, URL content | $0.01 + fee/query | | `blockrun_search` | Grok Live Search — web + X/Twitter + news with citations | $0.025 × max_results | | `blockrun_dex` | Live DEX prices via DexScreener | free | -| `blockrun_rpc` | Raw JSON-RPC on 40 chains (Ethereum, Base, Solana, Bitcoin, Sui, NEAR, …) via Tatum | $0.002/call | -| `blockrun_defi` | DefiLlama — protocol TVL, chain TVL, yield pools (APY), token prices | $0.001–0.005/call | +| `blockrun_rpc` | Raw JSON-RPC on 40 chains (Ethereum, Base, Solana, Bitcoin, Sui, NEAR, …) via Tatum | $0.002 + fee/call | +| `blockrun_defi` | DefiLlama — protocol TVL, chain TVL, yield pools (APY), token prices | $0.001–0.005 + fee/call | | `blockrun_modal` | Isolated code execution in a BlockRun-hosted Modal sandbox — disposable container, optional GPU (T4 → H100) | $0.01 create; $0.001/op | | `blockrun_phone` | Outbound AI voice calls (Bland) + wallet-owned US/CA numbers (Twilio), carrier + fraud lookups | $0.54/call; $5/number | | `blockrun_models` | Live catalogue of every LLM/image/video/music model + pricing | free | | `blockrun_wallet` | Balance, spending, agent budgets, setup QR, chain switch | free | +Flat data prices are the **base**. In wallet mode the gateway adds its flat network fee on top — $0.001 per call on Base today, and the Solana gateway quotes the base alone; the account rail charges no fee. The exact figure is in the `payment-required` header of any unpaid request, which is free to ask for. The server reserves $0.002 for the fee against the budget cap, so `blockrun_wallet action:"report"` and the spend-confirmation dialog run $0.001 high per call by design. + --- ## Key use cases 1. **Prediction-market consensus** → *"Polymarket's odds for the next Fed decision?"* — `blockrun_markets` 2. **Signal → trade** *(the full loop, self-custody)* → *"If 'hold' is under 30%, put $2 on Yes."* — `blockrun_markets` reads, `blockrun_polymarket action:"buy"` places. Gasless, confirm-gated. -3. **On-chain forensics** → *"This wallet — what's it labeled, what does it hold, when did it whale up?"* — `blockrun_surf` +3. **Smart-money forensics** → *"This Polymarket whale — who are they, which wallets are theirs, what's their P&L?"* — `blockrun_markets` `polymarket/wallet/identity/:wallet` + `.../cluster` 4. **Cited research** → *"5 most-cited papers on speculative decoding, last 90 days."* — `blockrun_exa` 5. **Image generation with on-image text** → *"Poster announcing GPT-5.5, retro-futuristic, headline 'NOW LIVE'."* — `blockrun_image` 6. **Give your agent a voice** → *"Speak this with the sarah voice."* — `blockrun_speech` 7. **Voice phone-out** → *"Call +1-415-… and confirm Friday at 3pm."* — `blockrun_phone` 8. **Multi-agent research, capped** → *"Spawn 3 agents on competing L1 narratives. Cap each at $0.50."* — `blockrun_wallet delegate × 3` -9. **Cross-chain SQL** → *"Top 10 tokens by DEX volume on Base, last 24h."* — `blockrun_surf` `onchain/sql` +9. **Raw chain reads, 40 chains** → *"Latest Base block, and this contract's USDC balance."* — `blockrun_rpc` --- @@ -496,7 +499,7 @@ Almost everything now settles on either chain. The exceptions: | Capability | API key | Solana wallet | Base wallet | |---|:--:|:--:|:--:| | Chat, image, video, music, speech, RealFace | ✅ | ✅ | ✅ | -| Search, Exa, Surf, markets, RPC, DEX, phone | ✅ | ✅ | ✅ | +| Search, Exa, markets, RPC, DEX, phone | ✅ | ✅ | ✅ | | `blockrun_defi` (DefiLlama) | ✅ | ❌ not served on the Solana gateway | ✅ | | `blockrun_modal` (sandboxes) | ✅ | ❌ not configured on the Solana gateway | ✅ | | Native Anthropic `claude-*` passthrough | ✅ | ❌ the SDK signs EIP-3009 only | ✅ | @@ -512,7 +515,7 @@ A blocked capability returns a message naming the fix, not a raw error. |---|---| | API key — most tools | **The amount actually settled**, read from the account API's per-call response | | API key — `blockrun_chat` | An estimate. Chat settles *after* the response by design, so no figure exists when the answer is sent | -| API key — `blockrun_image`, paid `blockrun_price` | An estimate, until the SDK surfaces the settled figure on those paths | +| API key — `blockrun_image` | **The amount actually settled** (since 0.49.0); only when the account API returns no figure does it fall back to the catalog estimate, marked `~` | | Wallet | The amount signed and settled on-chain, from the 402 quote | Anything estimated is printed with a `~` and says so. Estimates run **high** on @@ -528,7 +531,6 @@ cap trips early rather than late. The invoice is always - **CRITICAL: On any payment / balance / 402 error, call `blockrun_wallet` *first*** to check status, then `action:"setup"` for funding. Don't retry the failing tool blindly — the wallet is empty. - **CRITICAL: `blockrun_polymarket` moves REAL user funds** (pUSD on Polygon), separate from the x402 API budget. Never `buy`/`sell`/`redeem` with `confirm:true` unless the user explicitly approved that exact trade; without `confirm` you get a safe dry-run. Discover markets/token IDs with `blockrun_markets` first. -- **CRITICAL: `blockrun_surf`'s 84-endpoint catalog is in [`skills/surf/SKILL.md`](skills/surf/SKILL.md); `blockrun_markets`' full endpoint list is in its tool description** (worked examples in [`skills/prediction-markets/SKILL.md`](skills/prediction-markets/SKILL.md); live-demo workflow in [`skills/signal-to-trade-demo/SKILL.md`](skills/signal-to-trade-demo/SKILL.md)). Browse those before guessing paths. - **CRITICAL: `blockrun_music` and `blockrun_video` are payment-on-completion async.** Failures / client timeouts do NOT charge. Don't retry-loop — they may take 60–180s. - **CRITICAL: Before spawning child agents, allocate per-agent budget:** `blockrun_wallet action:"delegate" agent_id:"X" agent_limit:1.00`, then pass `agent_id:"X"` to every downstream call. The child is auto-blocked at zero. - **Free tier first for drafts:** `blockrun_chat mode:"free"` (NVIDIA), `blockrun_dex`, `blockrun_price` (crypto/FX/commodity), and `blockrun_models` are $0. @@ -557,9 +559,9 @@ Prompts and a worked example are in [`skills/image-prompting/SKILL.md`](skills/i | | Direct APIs | BlockRun | |---|---|---| -| Exa | Sign up, $20/mo minimum | $0.01/call, no subscription | -| Polymarket | Undocumented, rate-limited | $0.0095/call, clean JSON — plus you can **trade** | -| Surf (asksurf.ai) | Account + monthly plan | $0.0095/call, no account, 83 endpoints | +| Exa | Sign up, $20/mo minimum | $0.011/call on Base ($0.01 + fee), no subscription | +| Polymarket | Undocumented, rate-limited | $0.0085/call on Base ($0.0075 + fee), clean JSON — plus you can **trade** | +| DefiLlama | Free tier, rate-limited, no SLA | $0.006/call on Base ($0.005 + fee), same JSON, one wallet | | Multiple sources | 3 accounts, 3 API keys, 3 billing pages | **1 wallet** | One wallet. All sources. No dashboards. @@ -619,6 +621,8 @@ The server runs a non-blocking npm registry check at startup and prints an `Upda - **`claude mcp list` doesn't show `blockrun`** → Check `node -v` (≥20.19). Clear the npx cache: `rm -rf ~/.npm/_npx`. Re-run the install. - **`fetch failed` / balance-check timeout** → Base RPC transient outage. The tool falls through 3 public RPCs; retry after 30s. Persistent = local proxy / firewall blocking outbound RPC. - **`Video`/`Music generation timed out`** → Upstream queue congestion. **No charge** (payment-on-completion). Retry, or pick a faster model. +- **`blockrun_price` says `Equity quotes are not served (gateway 501 …)`** → Equity price/history were withdrawn on 2026-09-05; not an outage, and **nothing was charged** (the wallet is never asked to sign). The ticker catalog (`action:"list" category:"stocks"`) is still free. Equity coverage: hello@blockrun.ai. +- **`blockrun_markets` on `sports/*` fails — before 0.49.0 as `API error after payment: 502` with no balance change** → Predexon's `sports/*` routes have been down upstream since 2026-08-04; the gateway releases the payment on that upstream 500, so the call is **not charged** (the error says so when the gateway's "payment NOT charged" confirmation is in the response; otherwise it tells you to check `blockrun_wallet action:"report"`). For sports odds use `path:"markets/search"` with `params:{ q: "NBA" }`, or `polymarket/events` with `params:{ search: "NBA" }` — the bare `markets` route and its `league` filter were removed upstream on 2026-08-04 and 404 before payment. Upgrade to ≥ 0.49.0 so the error says all of this itself. - **No spend-confirmation dialog although `BLOCKRUN_CONFIRM_SPEND=on`** → Your client doesn't support MCP elicitation (Windsurf, Codex, Gemini CLI); the server proceeds without asking by design. Use `BLOCKRUN_BUDGET_LIMIT` as the guard, or a client from the [support table](#%EF%B8%8F-human-in-the-loop-payments). - **Polymarket: neg-risk ("winner") market buy fails, or `redeem` reverts, though setup shows ready** → Re-run `action:"setup" confirm:true` once (grants the on-chain approvals a pre-upgrade deposit wallet may lack — including the collateral-adapter approvals `redeem` needs). See the [setup guide](docs/polymarket-trading-setup.md). @@ -627,7 +631,7 @@ The server runs a non-blocking npm registry check at startup and prints an `Upda ## FAQ **What is BlockRun MCP?** -An open-source MCP server that gives Claude and other agents 20 tools for real-time data and real actions (trading, media, on-chain), paid per call — from a self-custody wallet or a BlockRun account key. +An open-source MCP server that gives Claude and other agents 19 tools for real-time data and real actions (trading, media, on-chain), paid per call — from a self-custody wallet or a BlockRun account key. **Do I need an API key or an account?** No. A wallet is auto-created locally on first run; you fund it with USDC and there are no signups, dashboards or keys to rotate. @@ -664,7 +668,7 @@ Both. Switch instantly with `blockrun_wallet action:"chain"`. A few media/paid t BlockRun is agent-native AI infrastructure — one wallet, x402 USDC micropayments, across every surface: -- **⚡ [ClawRouter](https://github.com/BlockRunAI/ClawRouter)** — the agent-native LLM router for OpenClaw. 76 models, <1ms local routing, USDC on Base & Solana. +- **⚡ [ClawRouter](https://github.com/BlockRunAI/ClawRouter)** — the agent-native LLM router for OpenClaw. 78 models, <1ms local routing, USDC on Base & Solana. - **🤖 [BRCC](https://blockrun.ai/brcc.md)** — BlockRun for Claude Code: smart routing + x402 payments, purpose-built for Claude Code. - **🐍 [ClawRouter-Hermes](https://github.com/BlockRunAI/ClawRouter-Hermes)** — Python plugin wiring NousResearch Hermes into the ClawRouter proxy. - **📚 [Docs](https://blockrun.ai/docs)** · **[Models & pricing](https://blockrun.ai/models)** — full SDKs, APIs, and the model catalogue. diff --git a/apps/order-preview.ts b/apps/order-preview.ts index 46be763..87a67ce 100644 --- a/apps/order-preview.ts +++ b/apps/order-preview.ts @@ -24,6 +24,8 @@ interface Preview { outcome?: string; conditionId?: string; bestQuote?: number | null; + /** Market orders: the limit the order will be signed at (buy: max, sell: min price per share). */ + worstFillPrice?: number; minSize?: number; maxBetUsd?: number; sessionSpentUsd?: number; @@ -114,6 +116,11 @@ function renderPreview(p: Preview): void { const grid = el("div", { class: "grid" }, kv(isBuy ? "You spend" : "You receive (est.)", usd(p.notionalUsd), true), kv(priceLabel, `${prob} · ${Number.isFinite(priceVal) ? priceVal.toFixed(3) : "—"}`, true), + // Market orders are SIGNED at this bound (the server walks the book), so + // the fill can never be worse than the number shown here. + ...(!isLimit && typeof p.worstFillPrice === "number" + ? [kv(isBuy ? "Worst fill (signed max)" : "Worst fill (signed min)", `${(p.worstFillPrice * 100).toFixed(1)}¢ · ${p.worstFillPrice.toFixed(3)}`)] + : []), kv("Shares", shares !== undefined ? `${isLimit ? "" : "≈ "}${shares.toFixed(4)}` : "—"), kv("Max payout if right", isBuy && shares !== undefined ? usd(shares) : "—"), kv("Per-order cap", el("span", {}, `${usd(p.notionalUsd)} of ${cap ? usd(cap) : "—"}`, el("div", { class: "meter" }, el("i", { style: `width:${capPct}%` })))), @@ -169,7 +176,25 @@ function renderPreview(p: Preview): void { let armed = false; const disarm = () => { armed = false; place.textContent = `Place ${p.action} · ${usd(p.notionalUsd)}`; place.classList.remove("danger"); cancel.hidden = true; }; cancel.addEventListener("click", disarm); + + // Every figure on this card — notional, shares, worst fill, the confirm + // label — describes the amount that was QUOTED. currentArgs() reads the + // field live, so an edited amount used to be submitted under the old + // label ("Confirm — sign & submit $5.00" placing $50). Placing is only + // allowed while the field still equals the quoted amount; a change disarms + // and disables Place until Re-quote renders a fresh card. + const quotedAmount = parseFloat(amountField.value); + const syncPlace = () => { + const stale = parseFloat(amountField.value) !== quotedAmount; + if (stale && armed) disarm(); + place.disabled = stale; + place.title = stale ? "Amount changed — Re-quote first" : ""; + if (stale) { note.className = "note"; note.textContent = "Amount changed — Re-quote first to refresh the price and notional before placing."; } + }; + amountField.addEventListener("input", syncPlace); + place.addEventListener("click", async () => { + if (parseFloat(amountField.value) !== quotedAmount) { syncPlace(); return; } if (!armed) { armed = true; place.textContent = `Confirm — sign & submit ${usd(p.notionalUsd)}`; diff --git a/assets/context-cost-dark.svg b/assets/context-cost-dark.svg index 747163c..015e3e4 100644 --- a/assets/context-cost-dark.svg +++ b/assets/context-cost-dark.svg @@ -1,10 +1,10 @@ - + CONTEXT COST - 13.0K tokens + 12.7K tokens 6% of a 200K context window · every turn, whether or not you call a tool - 5.6K with --profile trading — 57% less + 5.2K with --profile trading — 59% less measured, not estimated diff --git a/assets/context-cost.svg b/assets/context-cost.svg index 5d4b729..665256a 100644 --- a/assets/context-cost.svg +++ b/assets/context-cost.svg @@ -1,10 +1,10 @@ - + CONTEXT COST - 13.0K tokens + 12.7K tokens 6% of a 200K context window · every turn, whether or not you call a tool - 5.6K with --profile trading — 57% less + 5.2K with --profile trading — 59% less measured, not estimated diff --git a/brand-numbers.json b/brand-numbers.json index 28b0d60..90d551d 100644 --- a/brand-numbers.json +++ b/brand-numbers.json @@ -2,17 +2,17 @@ "$schema": "https://blockrun.ai/brand/numbers.schema.json", "version": 1, "models": { - "chatVisible": 76, - "totalVisible": 100, - "free": 7, - "freeWithheld": 26, + "chatVisible": 78, + "totalVisible": 102, + "free": 6, + "freeWithheld": 27, "image": 9, "video": 8, "music": 1, "speech": 5, "soundfx": 1, - "withFallback": 34, - "withFallbackAllEntries": 73 + "withFallback": 33, + "withFallbackAllEntries": 74 }, "clawrouter": { "dimensions": 15, @@ -21,7 +21,7 @@ "aliases": 259 }, "mcp": { - "tools": 20, + "tools": 19, "contextTokens": 12900, "contextTokensTrading": 5554, "contextCutPct": 57 diff --git a/docs/polymarket-trading-setup.md b/docs/polymarket-trading-setup.md index 06fdf55..6376a5c 100644 --- a/docs/polymarket-trading-setup.md +++ b/docs/polymarket-trading-setup.md @@ -6,9 +6,11 @@ AI via x402. One self-custody identity: it pays for models in USDC on Base *and* settles USDC-denominated bets on the world's largest prediction market. > **Real money.** A confirmed order spends real **pUSD** (Polymarket's USDC-backed -> collateral) on Polygon. Every order, approval, redeem, and withdrawal is -> **confirm-gated** (dry-run unless you pass `confirm:true`) and **capped** -> (`POLYMARKET_MAX_BET_USD`, default **$25/order**). Start with ~$5 and a $1 test. +> collateral) on Polygon. Every order, approval, fund, redeem, and withdrawal is +> **confirm-gated** (dry-run unless you pass `confirm:true`). Orders are +> additionally **capped** (`POLYMARKET_MAX_BET_USD`, default **$25/order**; +> optional `POLYMARKET_MAX_SESSION_USD`); funding can be capped per call with +> the optional `POLYMARKET_MAX_FUND_USD` (unset = no cap). Start with ~$5 and a $1 test. This flow is verified end to end on the live CLOB: an agent's own wallet created its deposit vault, funded it gaslessly via x402, and placed a **real $1 market @@ -199,7 +201,8 @@ blockrun_polymarket action:"redeem" condition_id:"0x..." confirm:true blockrun_polymarket action:"withdraw" # dry-run (full balance) blockrun_polymarket action:"withdraw" confirm:true # execute # partial: amount_usd:5 -# elsewhere: to_address:"0x..." (default: your own agent wallet on Base) +# elsewhere: to_address:"0x..." (default: your own agent wallet on Base; +# a custom address is flagged CUSTOM in the preview — read it twice) ``` **The loop that closes the story:** `sell` (before resolution) or `redeem` @@ -216,6 +219,12 @@ x402 AI fees. Money in via x402, money out via withdraw — one wallet, full cir never signs *above* your limit, a sell never *below*. - **Market buy** takes `amount_usd` (dollars to spend); **market sell** takes `size` (shares to sell). **Limit** takes `price` + `size`. +- **Market orders are signed at the previewed worst fill.** The dry-run walks + the live book and prints `worst fill ≤ X` (buy) / `worst fill ≥ X` (sell) + next to the best quote; that same `X` is the limit on the signed order, so a + book that thins between preview and confirm can only fill *less*, never + worse than what you saw. Est. sell proceeds are the walked total, not + size × best bid. - **Order types:** `GTC` (rests, default for limits), `GTD` (good-till-`expires_at`), `FOK` (fill-or-kill, default for market), `FAK` (fill-and-kill the rest). - `min_order_size` and tick come from the live market — the dry-run shows both. @@ -227,7 +236,17 @@ x402 AI fees. Money in via x402, money out via withdraw — one wallet, full cir - **`confirm:true` is required** for every order / approval / fund / redeem / withdraw. Without it: a dry-run preview, nothing signed. - **Per-order cap** `POLYMARKET_MAX_BET_USD` (default $25) + optional cumulative - **`POLYMARKET_MAX_SESSION_USD`** (in-memory, per-`agent_id`). + **`POLYMARKET_MAX_SESSION_USD`** (in-memory, per-`agent_id`). These gate + orders only. +- **Optional per-call fund cap** `POLYMARKET_MAX_FUND_USD` — bounds a single + `fund` call (your own Base USDC → your own vault). Unset = no cap, which is + the default: funding is self-to-self and reversible via `withdraw`. `redeem` + and `withdraw` are confirm-gated but not capped. +- **`withdraw` labels the destination honestly.** The preview says + `(your agent wallet)` only when the money is going back to the wallet that + pays your x402 fees; a caller-supplied `to_address` is shown as + `⚠️ CUSTOM destination — NOT your agent wallet`, and a malformed or + checksum-mismatched address is refused before anything is signed. - **Bets never draw from your x402 API budget** — different asset (pUSD vs Base USDC), different wallet ledger. They can't corrupt each other. - **Your private key never leaves the machine** and is never printed or logged. diff --git a/package-lock.json b/package-lock.json index 2ba618b..7dd42a7 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,16 +1,16 @@ { "name": "@blockrun/mcp", - "version": "0.45.1", + "version": "0.49.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@blockrun/mcp", - "version": "0.45.1", + "version": "0.49.0", "license": "MIT", "dependencies": { "@anthropic-ai/sdk": "^0.123.0", - "@blockrun/llm": "^3.14.3", + "@blockrun/llm": "^3.15.1", "@modelcontextprotocol/sdk": "^1.0.0", "@polymarket/builder-relayer-client": "^0.0.10", "@polymarket/builder-signing-sdk": "1.0.0", @@ -84,20 +84,18 @@ } }, "node_modules/@blockrun/llm": { - "version": "3.14.3", - "resolved": "https://registry.npmjs.org/@blockrun/llm/-/llm-3.14.3.tgz", - "integrity": "sha512-bJD4Fp8hiXuCcuQbPaSRO8eRtPPAdfGcjVoMQP457qjfAhhN/Xwz59uEiTIBnMyIx5hRW1UXepxSU3m4sMuUbA==", + "version": "3.15.1", + "resolved": "https://registry.npmjs.org/@blockrun/llm/-/llm-3.15.1.tgz", + "integrity": "sha512-Q8IyUrqXPnBfIlGCyzJN3xj4ahgEoVXs6jq+K0qL89DR1zYUljDwic32Yr8uvK7bfkQ6EdY4mXBHcMDx7EaApg==", "license": "MIT", "dependencies": { + "@anthropic-ai/sdk": "^0.123.0", "bs58": "^6.0.0", "viem": "^2.56.3" }, "engines": { "node": ">=20" }, - "optionalDependencies": { - "@anthropic-ai/sdk": "^0.123.0" - }, "peerDependencies": { "@solana/spl-token": "^0.4.15", "@solana/web3.js": "^1.98.4" diff --git a/package.json b/package.json index 0e3d674..d88d833 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@blockrun/mcp", - "version": "0.48.0", + "version": "0.49.0", "mcpName": "io.github.BlockRunAI/blockrun-mcp", "description": "BlockRun MCP Server - Give your AI agent web search, deep research, prediction markets, and crypto data. Pay per call from a USDC wallet (Solana or Base) or a BlockRun API key.", "type": "module", @@ -58,7 +58,7 @@ }, "dependencies": { "@anthropic-ai/sdk": "^0.123.0", - "@blockrun/llm": "^3.14.3", + "@blockrun/llm": "^3.15.1", "@modelcontextprotocol/sdk": "^1.0.0", "@polymarket/builder-relayer-client": "^0.0.10", "@polymarket/builder-signing-sdk": "1.0.0", diff --git a/scripts/verify-prices.ts b/scripts/verify-prices.ts index add76a0..0220db3 100644 --- a/scripts/verify-prices.ts +++ b/scripts/verify-prices.ts @@ -1,6 +1,8 @@ // scripts/verify-prices.ts — run with: npm run verify:prices // -// Compares every local cost estimator against what the LIVE gateway quotes. +// Compares every local cost estimator against what the LIVE gateway quotes, and +// the chat price table against the live model catalogue (see the sweep at the +// end — it is the only check that can see a model the table does NOT list). // // WHY: the estimators feed the budget gate. If one under-quotes, the gate // reserves less than the call settles for and an agent walks past its cap; the @@ -9,7 +11,8 @@ // silent, and by construction it appears in production, not in CI. // // This has already gone wrong three times in the same direction: -// 1. stale Surf tiers after the gateway went flat, +// 1. stale Surf tiers after the gateway went flat (Surf itself is gone since +// 2026-09-06 — kept in this list because the defect class is not), // 2. the 402 body's `price` (the BASE) mistaken for what x402 charges, // 3. round() instead of the gateway's ceil(), one micro short wherever a // x1.05 margin drifts in float. @@ -23,7 +26,6 @@ // as "free" rather than raising. Hence the explicit check below. import { estimateModalCost } from "../src/tools/modal.js"; import { estimatePhoneCost } from "../src/tools/phone.js"; -import { estimateSurfCost, SURF_PRICE_USD } from "../src/tools/surf.js"; import { estimateSearchCost } from "../src/tools/search.js"; import { estimateCost as estimateImageCost } from "../src/tools/image.js"; import { estimateExaCost } from "../src/tools/exa.js"; @@ -31,6 +33,7 @@ import { estimateChatCost, promptCharSize } from "../src/tools/chat.js"; import { estimateVideoCost } from "../src/tools/video.js"; import { MARKETS_PRICE_USD } from "../src/tools/markets.js"; import { withTxFee } from "../src/utils/tx-fee.js"; +import { CHAT_PRICE_PER_MTOKEN, DEFAULT_CHAT_PRICE, FREE_CHAT_MODELS, MODEL_TIERS } from "../src/utils/constants.js"; // TWO gateways, and they do not agree. Base and Solana are separate deployments // with separate env, and TRANSACTION_FEE_USD is env-overridable in the gateway — @@ -61,14 +64,16 @@ type Probe = { allowOver?: boolean; }; -async function quote(host: string, path: string, body?: unknown): Promise { +type Quote = { usd: number; description?: string }; + +async function quote(host: string, path: string, body?: unknown): Promise { const res = await fetch(host + path, { method: body === undefined ? "GET" : "POST", ...(body === undefined ? {} : { headers: { "content-type": "application/json" }, body: JSON.stringify(body) }), }); const header = res.headers.get("payment-required"); if (!header) return `no 402 (HTTP ${res.status})`; - let parsed: { accepts?: Array<{ amount?: string }> }; + let parsed: { accepts?: Array<{ amount?: string; extra?: { description?: string } }>; resource?: { description?: string } }; try { parsed = JSON.parse(Buffer.from(header.trim(), "base64").toString("utf8")); } catch { @@ -78,15 +83,20 @@ async function quote(host: string, path: string, body?: unknown): Promise { + // Every model priced ABOVE the $5/$30 default an unknown model falls back to + // (gpt-5.4-pro is the `powerful` row above). Reachable as an explicit + // `model`, where no tier bound applies at all. This list is hand-written, and + // that is exactly how gpt-6-astra and claude-fable-5.1 sat unprobed for weeks + // after landing at $10/$50 — the catalogue sweep below is what catches the + // next one; this list only pins the reserve for the ones already known. + ...(["openai/gpt-5.5-pro", "openai/gpt-5.2-pro", "openai/o1", "anthropic/claude-fable-5", "openai/gpt-6-astra", "anthropic/claude-fable-5.1"].map((model) => { const message = "word ".repeat(20_000); return { label: `chat explicit ${model.split("/")[1]} 100k`, @@ -273,16 +287,22 @@ let solShort = 0; // reserve < what SOLANA charges -> genuine under-reserve, blo let solDearer = 0; // Solana dearer than Base but still covered -> policy note only let solCheaper = 0; // Solana charges LESS than Base -> safe, but the docs quote one number let solMissing = 0; // route not served on Solana at all +let solSubstituted = 0; // Solana quotes a DIFFERENT product than Base for the same request const solNotes: string[] = []; +const solSubstitutions: string[] = []; console.log(`Verifying ${PROBES.length} routes against live 402 quotes on BOTH gateways (free — no payment attached)\n`); for (const probe of PROBES) { // Both chains at once so adding the second gateway costs no wall-clock. - const [live, solLive] = await Promise.all([ + const [liveQ, solQ] = await Promise.all([ quote(BASE, probe.path, probe.body), quote(SOL, probe.path, probe.body), ]); + const live = typeof liveQ === "string" ? liveQ : liveQ.usd; + const solLive = typeof solQ === "string" ? solQ : solQ.usd; + const liveProduct = typeof liveQ === "string" ? undefined : product(liveQ.description); + const solProduct = typeof solQ === "string" ? undefined : product(solQ.description); // Compare the chains before judging the estimator, so a Solana-only problem is // still reported when the Base probe itself is unreachable. @@ -291,6 +311,27 @@ for (const probe of PROBES) { solMissing++; solTag = " [sol: not served]"; solNotes.push(`${probe.label}: Solana ${solLive}`); + } else if ( + typeof live === "number" && liveProduct && solProduct && liveProduct !== solProduct && + // A different label alone is not a substitution — the Solana gateway writes + // longer marketing descriptions for the same route (rpc/ethereum: 40 words + // vs Base's 5, same $0.002). Substitution is a different label AND a price + // materially ABOVE Base for the same request; a cheaper Solana row falls + // through to the deliberate-pricing branch below. + solLive - live > Math.max(EPSILON, 0.25 * live) + ) { + // Not a price for the same thing. Found 2026-09-08: sol.blockrun.ai (a + // separate deployment that can lag Base) did not know azure/sora-2 and + // quoted "Seedance 2.0 Pro video generation (5s)" at $1.135480 in its + // place. That is a gateway bug to report, not an estimator gap to paper + // over by reserving the substitute's price — and since 0.49.0 every + // manual-402 tool refuses a quote this far off the published rate before + // signing (assertQuoteNearEstimate), nothing can be charged for it. Loud, + // but not a release blocker for this repo. + solSubstituted++; + const guarded = /^(videos|images)\//.test(probe.path); + solTag = ` [sol: quotes a DIFFERENT product — "${typeof solQ === "string" ? "" : solQ.description}" at $${solLive.toFixed(6)}; ${guarded ? "the tool refuses it unsigned" : "NOT guarded — this tool reserves the Base figure"}]`; + solSubstitutions.push(`${probe.label}: Base sells "${typeof liveQ === "string" ? "" : liveQ.description}" at $${live.toFixed(6)}, Solana sells "${typeof solQ === "string" ? "" : solQ.description}" at $${solLive.toFixed(6)}`); } else if (typeof live === "number") { const chainDelta = solLive - live; if (chainDelta > EPSILON) { @@ -346,8 +387,13 @@ console.log( if (unreachable) console.log("Unreachable routes were NOT verified — treat them as unknown, not as passing."); console.log( - `Solana: ${solShort} under-reserved (BLOCKER), ${solDearer} dearer than Base but covered, ${solCheaper} cheaper, ${solMissing} not served`, + `Solana: ${solShort} under-reserved (BLOCKER), ${solDearer} dearer than Base but covered, ${solCheaper} cheaper, ${solMissing} not served, ${solSubstituted} substituted`, ); +if (solSubstituted) { + console.log(" GATEWAY BUG — Solana quotes a different, dearer product than Base for the same request. blockrun_video"); + console.log(" and blockrun_image refuse such a quote before signing (assertQuoteNearEstimate); report it to the gateway owner:"); + for (const n of solSubstitutions) console.log(` ${n}`); +} if (solCheaper) { console.log( " Solana charges no transaction fee — DELIBERATE pricing (owner decision,\n" + @@ -364,13 +410,119 @@ if (solMissing) { console.log(" Not served on Solana — an agent that switched chains gets a 404/503, not a cheaper call:"); for (const n of solNotes) console.log(` ${n}`); } +// ---- CATALOGUE SWEEP ---- +// +// Everything above checks rows the chat price table HAS. This checks the rows it +// LACKS. An explicit `model` with no CHAT_PRICE_PER_MTOKEN row reserves +// DEFAULT_CHAT_PRICE, which is only safe while nothing in the catalogue is priced +// above it — a premise the table's header asserted and nothing verified. It was +// false for weeks: openai/gpt-6-astra and anthropic/claude-fable-5.1 landed at +// $10/$50 on both gateways with no row, so the gate reserved half of what +// settled, and the 402 probes above never saw them because they only probe ids +// someone thought to list. GET /v1/models is free and unauthenticated: read it +// and fail on any available chat model the reserve does not cover. +// +// A row that exists but reads BELOW the live rate is the same under-reserve with +// a different cause (a reprice rather than a new model) and fails the same way. +// A FREE_CHAT_MODELS member that the catalogue now PRICES is the worst case of +// all — the gate reserves $0 for it — and fails too. +// +// Rows ABOVE the live rate are the safe direction on the OpenAI-compat paths +// (over-reserve; the ledger books the real settle) and only warn — EXCEPT for +// anthropic/* on Base, where the native /v1/messages path has no settlement +// counter and anthropicCallCost books THIS TABLE. claude-sonnet-5 sat at $3/$15 +// for weeks after both gateways cut it to $2/$10: a 1.5x over-count on every +// call, tripping caps at two-thirds of their allowance. That fails. +// +// Listed-but-unknown $0 models and unlisted free[] entries are reported, not +// failed: the first only over-reserves, and absence from the catalogue is a +// listing decision, not a death certificate (see the doctrine in constants.ts). +type CatalogueModel = { id: string; available?: boolean; pricing?: { input?: unknown; output?: unknown } }; + +async function catalogue(host: string): Promise { + try { + const res = await fetch(host + "models"); + if (!res.ok) return `HTTP ${res.status}`; + const body = (await res.json()) as { data?: unknown }; + return Array.isArray(body.data) ? (body.data as CatalogueModel[]) : "no `data` array in the response"; + } catch (err) { + return err instanceof Error ? err.message : String(err); + } +} + +const catalogueGaps: string[] = []; // fail +const catalogueNotes: string[] = []; // report only +let catalogueUnreachable = 0; +console.log("\nCatalogue sweep: every live chat model must be covered by its price row, by the default, or by FREE_CHAT_MODELS"); +for (const [name, host] of [["Base", BASE], ["Solana", SOL]] as const) { + const models = await catalogue(host); + if (typeof models === "string") { + console.log(` ? ${name.padEnd(26)} ${models}`); + catalogueUnreachable++; + continue; + } + let checked = 0; + let gaps = 0; + const listed = new Set(); + for (const m of models) { + const { input, output } = m.pricing ?? {}; + // Per-image, per-second and per-character products share the catalogue but + // not this price table; only $/M-token pricing is a chat model. + if (typeof input !== "number" || typeof output !== "number") continue; + // Base marks retired rows `available:false`; Solana omits the field + // entirely, and an omitted flag is a served model, not an unknown one. + if (m.available === false) continue; + checked++; + listed.add(m.id); + const isFree = FREE_CHAT_MODELS.has(m.id); + const row = Object.hasOwn(CHAT_PRICE_PER_MTOKEN, m.id) ? CHAT_PRICE_PER_MTOKEN[m.id] : undefined; + // What estimateChatCost reserves for an explicit call to this id. + const reserve = isFree ? { input: 0, output: 0 } : (row ?? DEFAULT_CHAT_PRICE); + if (input > reserve.input || output > reserve.output) { + gaps++; + catalogueGaps.push( + `${name}: ${m.id} is $${input}/$${output} live but ` + + (isFree + ? "FREE_CHAT_MODELS lists it as free — the gate reserves $0 for a paid call" + : row + ? `its row reserves $${row.input}/$${row.output}` + : `has NO row and reserves the $${DEFAULT_CHAT_PRICE.input}/$${DEFAULT_CHAT_PRICE.output} default`), + ); + continue; + } + if (row && (input < row.input || output < row.output)) { + if (name === "Base" && m.id.startsWith("anthropic/")) { + // Native Anthropic is Base-only and books this row as the ledger. + gaps++; + catalogueGaps.push(`${name}: ${m.id} row is $${row.input}/$${row.output} but the gateway charges $${input}/$${output} — the native ledger over-books every call`); + } else { + catalogueNotes.push(`${name}: ${m.id} row $${row.input}/$${row.output} is above the live $${input}/$${output} — over-reserves (safe), but stale`); + } + } + if (!isFree && input === 0 && output === 0) { + catalogueNotes.push(`${name}: ${m.id} is billed $0 but FREE_CHAT_MODELS does not list it — an explicit call reserves the default, and an exhausted budget refuses a free call`); + } + } + for (const id of MODEL_TIERS.free) { + if (!listed.has(id)) catalogueNotes.push(`${name}: free[] routes ${id}, which the catalogue does not list — not a death certificate (gpt-oss-120b is hidden-alive); probe with a realistic POST before removing`); + } + console.log(` ${gaps ? "✗" : "✓"} ${name.padEnd(26)} ${checked} chat models checked, ${gaps} would settle above the reserve`); +} +for (const g of catalogueGaps) console.log(` ✗ ${g}`); +for (const n of catalogueNotes) console.log(` ! ${n}`); +if (catalogueUnreachable) console.log(" A catalogue that could not be read was NOT verified — treat it as unknown, not as passing."); + // Under-reserving is a release blocker: it means the budget cap is a lie. That is // true per CHAIN — an estimator built off Base is a lie on Solana the moment -// Solana costs more, and nothing else in the repo would notice. +// Solana costs more, and nothing else in the repo would notice. It is equally +// true for a catalogue model the table does not know: the gate reserves the +// default for it, and the default is a claim about the catalogue. // Over-reserving only blocks affordable calls, so it warns without failing. -if (short || solShort) { - console.log( - `\nFAIL: an estimator reserves less than the gateway charges${solShort ? " (on Solana)" : ""}. Fix it before publishing.`, - ); +if (short || solShort || catalogueGaps.length) { + const why = [ + short || solShort ? `an estimator reserves less than the gateway charges${solShort ? " (on Solana)" : ""}` : "", + catalogueGaps.length ? `${catalogueGaps.length} live chat model${catalogueGaps.length === 1 ? "" : "s"} disagree${catalogueGaps.length === 1 ? "s" : ""} with the price table in a direction that costs money` : "", + ].filter(Boolean).join("; "); + console.log(`\nFAIL: ${why}. Fix it before publishing.`); process.exit(1); } diff --git a/server.template.json b/server.template.json index 3e7b6e0..61e28f2 100644 --- a/server.template.json +++ b/server.template.json @@ -17,11 +17,25 @@ }, "environmentVariables": [ { - "description": "Optional: Your wallet private key for USDC payments (hex for Base, bs58 for Solana). If not set, a wallet is auto-generated.", + "description": "Optional: EVM private key (0x-prefixed hex) for USDC payments on Base; also the Polymarket signer. If not set, a wallet is auto-generated.", "isRequired": false, "format": "string", "isSecret": true, "name": "BLOCKRUN_WALLET_KEY" + }, + { + "description": "Optional: Solana private key (bs58) for USDC payments on Solana. Setting it selects the Solana chain. If not set, a wallet is auto-generated.", + "isRequired": false, + "format": "string", + "isSecret": true, + "name": "SOLANA_WALLET_KEY" + }, + { + "description": "Optional: BlockRun account API key; bills your prepaid balance instead of a wallet (no per-call network fee). Takes priority over a wallet when set.", + "isRequired": false, + "format": "string", + "isSecret": true, + "name": "BLOCKRUN_API_KEY" } ] } diff --git a/skills/blockrun-debug/SKILL.md b/skills/blockrun-debug/SKILL.md index 0dd4431..756156a 100644 --- a/skills/blockrun-debug/SKILL.md +++ b/skills/blockrun-debug/SKILL.md @@ -1,6 +1,6 @@ --- name: blockrun-debug -description: "Use when the BlockRun MCP server (@blockrun/mcp) is installed but misbehaving — 'Failed to connect', spawn npx ENOENT, blockrun missing from claude mcp list, HTTP 402 / Insufficient balance, fetch failed, video or music timeouts, spend-confirmation dialogs not appearing, or a Polymarket buy/redeem failing after funding. Symptom → cause → fix, plus what never to do." +description: "Use when the BlockRun MCP server (@blockrun/mcp) is installed but misbehaving — 'Failed to connect', spawn npx ENOENT, blockrun missing from claude mcp list, HTTP 402 / Insufficient balance, fetch failed, video or music timeouts, a 501 'not served' error or 'API error after payment' while the balance never moved, spend-confirmation dialogs not appearing, or a Polymarket buy/redeem failing after funding. Symptom → cause → fix, plus what never to do." triggers: - "blockrun failed to connect" - "blockrun not working" @@ -10,6 +10,12 @@ triggers: - "blockrun 402" - "fetch failed blockrun" - "video generation timed out" + - "api error after payment" + - "equity quotes are not served" + - "sports markets 500" + - "501 not implemented" + - "refusing to sign it" + - "quoted a different price" - "polymarket buy failed" - "insufficient allowance" - "redeem reverts" @@ -59,8 +65,13 @@ re-added at user scope leaves a duplicate. Then, in the session: `blockrun_walle | Startup error "not a valid BlockRun API key" | `BLOCKRUN_API_KEY` is malformed. It deliberately fails loudly rather than silently spending USDC from a wallet instead. | Fix the value or unset it. | | `fetch failed` / balance-check timeout | Base RPC blip; the tool rotates through 3 public RPCs | Wait 30 s, retry once. Persistent → a local proxy/firewall is blocking outbound RPC. | | `Video`/`Music generation timed out` | Upstream queue. **Not charged** — payment settles on completion only. | Retry, or pick a faster model. Do not retry-loop; jobs take 60–180 s. | +| `blockrun_price` with `category:"stocks"` / `"usstock"` → `Equity quotes are not served (gateway 501 …)` | The gateway withdrew equity price/history on 2026-09-05 (licensing), and the tool answers before the wallet is consulted. Not an outage. **Not charged.** | Do not retry. `action:"list" category:"stocks" market:"us"` still returns the ticker catalog for free. Equity coverage: hello@blockrun.ai. | +| `blockrun_markets` on `sports/*` → `Predexon's sports/* routes have returned an upstream 500 … since 2026-08-04` (builds before 0.49.0: `API error after payment: 502 / Request failed` with no balance change) | Upstream Predexon outage since 2026-08-04. The gateway releases the payment on the upstream 500, so the old wording asserted a charge that never happened. **Not charged** when the response carries the gateway's `(payment NOT charged)` confirmation — the error then says so; without it the error tells you to check `blockrun_wallet action:"report"`. | Use `path:"markets/search"` with `params:{ q:"NBA" }` or `polymarket/events` with `params:{ search:"NBA" }`. Not `markets` + `league` — removed upstream 2026-08-04, 404s before payment. Do not retry `sports/*`. Upgrade to ≥ 0.49.0 so the error says this itself. | +| A config names `blockrun_surf` → the client reports an unknown tool | The tool was REMOVED in 0.49.0: the gateway has answered every Surf path with `410 endpoint_retired` since 2026-09-06, so it could only ever error. **Nothing is charged.** | Use `blockrun_price` / `blockrun_dex` / `blockrun_defi` / `blockrun_markets` / `blockrun_rpc`; the `surf` skill maps each former endpoint. Drop `blockrun_surf` from any allowlist. | +| `blockrun_video` / `blockrun_image` → `The gateway quoted $X for , but this tool expected about $Y (N.Nx the published rate) … Refusing to sign it — no charge was made` | The 402 price is far above the published rate: the gateway repriced the model, or a lagging deployment substituted another one. Live 2026-09-08: `sol.blockrun.ai` does not know `azure/sora-2` and quotes Seedance 2.0 Pro at $1.135 in its place. **Not charged** — the tool refuses before signing. | For Sora: `blockrun_wallet action:"chain" chain:"base"`. Otherwise pick another model or chain, and report the quote (the message names what the gateway labelled it) so the estimator or the gateway gets fixed. | +| Any tool → `The gateway does not serve this endpoint (501 Not Implemented)` | The route is withdrawn, not down. Before payment the message ends "nothing was charged"; after payment it tells you to check the ledger instead, because the formatter cannot know whether the nonce was released. | Do not retry. `blockrun_wallet action:"report"` shows whether the call settled. | | Model id 404s | Delisted upstream | `blockrun_models` for the live list. | -| Startup prints `🚨 WALLET PRIVATE KEY DETECTED IN CONFIG FILE` | The key was pasted into `~/.claude.json` (old hosted-auth flow) | Treat the key as compromised: move funds to a new wallet, remove it from the config. | +| Startup prints `🚨 WALLET PRIVATE KEY DETECTED IN CONFIG FILE` | A raw key sits somewhere it was never meant to be — a header, an `args` entry, the old hosted-auth field — in a client config file (`~/.claude.json`, Claude Desktop, Cursor, Windsurf). Since 0.49.0 the DOCUMENTED override `mcpServers.*.env.BLOCKRUN_WALLET_KEY` / `SOLANA_WALLET_KEY` does NOT trigger this banner; it gets a short plaintext-and-synced note instead. | Banner: treat the key as compromised — move funds to a new wallet, remove it from the config. Note: optional; prefer `~/.blockrun/.session` or `BLOCKRUN_KEYCHAIN=auto`. | | No spend-confirmation dialog with `BLOCKRUN_CONFIRM_SPEND=on` | Client doesn't support MCP elicitation (Windsurf, Codex, Gemini CLI) — the server proceeds without asking, by design | Use `BLOCKRUN_BUDGET_LIMIT` / `blockrun_wallet action:"delegate"` as the guard, or use Claude Code / Cursor / VS Code where the dialog renders. | | Dialog appears, user clicks OK, tool says "declined" | Only an explicit **Decline** stops a charge; Cancel/ESC proceeds. If it says declined, Decline was pressed. | Re-run the call; approve it. | | `Update available: vX → vY` on stderr | Informational | Switch to the `blockrun-upgrade` skill. | diff --git a/skills/blockrun-setup/SKILL.md b/skills/blockrun-setup/SKILL.md index 760e4a3..2a7fc16 100644 --- a/skills/blockrun-setup/SKILL.md +++ b/skills/blockrun-setup/SKILL.md @@ -63,7 +63,7 @@ For a JSON client with nvm/Homebrew Node, put the absolute `npx` path (`which np `command` — there is no `-e PATH` equivalent there. **Optional flags** (append after `@latest`): `--profile trading|research|media|chat` -exposes a smaller tool set so the client loads fewer schemas. Omit for all 20 tools. +exposes a smaller tool set so the client loads fewer schemas. Omit for all 19 tools. **Optional env** (`-e KEY=value` on Claude Code, `"env": {}` in JSON): `BLOCKRUN_CONFIRM_SPEND=on` asks before each paid call on clients that support MCP diff --git a/skills/blockrun/SKILL.md b/skills/blockrun/SKILL.md index 1fa4b64..68dfa4e 100644 --- a/skills/blockrun/SKILL.md +++ b/skills/blockrun/SKILL.md @@ -7,7 +7,7 @@ description: | question, how the wallet works, or how to make a first call for free. TOOLS: blockrun_chat, blockrun_image, blockrun_video, blockrun_music, blockrun_speech, blockrun_search, blockrun_exa, blockrun_markets, blockrun_polymarket_read, blockrun_polymarket, - blockrun_surf, blockrun_price, blockrun_dex, blockrun_defi, blockrun_rpc, blockrun_phone, + blockrun_price, blockrun_dex, blockrun_defi, blockrun_rpc, blockrun_phone, blockrun_realface, blockrun_modal, blockrun_models, blockrun_wallet. TRIGGERS: blockrun, x402, use grok, use gpt, use deepseek, compare models, generate image, generate video, generate music, text to speech, web search, news search, prediction market, @@ -34,7 +34,7 @@ changed, because each one was a typed copy. Read prices from a live source inste | What a call will actually cost | the `402` response — it carries the real amount | | Full endpoint catalog with prices | (Base) · (Solana) | -The tool descriptions in the MCP server carry current prices too; they are generated, not typed. +The tool descriptions in the MCP server carry the published base prices too; they are typed by hand and verified against live 402 quotes by `npm run verify:prices`, and the 402 header is what actually gets charged. ## The two chains are not the same gateway @@ -68,7 +68,7 @@ digits — an unmarked number is invisible to that check, which is exactly how t ## Getting a first call working -**Free, no wallet, no key.** 7 open-weight +**Free, no wallet, no key.** 6 open-weight chat models cost nothing. Use `blockrun_chat` with `mode: "free"` — the parameter is `mode`, not `routing`, and an unrecognised key is silently dropped, which lands you on the paid `balanced` tier instead. Calling the HTTP API directly, a free model needs no wallet and no @@ -111,7 +111,6 @@ Deeper wallet, budget and x402 mechanics — including calling the HTTP API dire | Token / FX / commodity price | `blockrun_price` | Crypto, FX and commodities are free | | DEX pairs and liquidity | `blockrun_dex` | Free | | DeFi TVL, yields | `blockrun_defi` | | -| On-chain SQL, wallet labels, social mindshare | `blockrun_surf` | The deep crypto tool | | Raw JSON-RPC against a chain | `blockrun_rpc` | 40 chains, one gateway | | Phone lookup, buy a number, make an AI call | `blockrun_phone` | Buy the number first | | Run code in a remote container / on a GPU | `blockrun_modal` | Prefer local for normal repo work | @@ -119,7 +118,9 @@ Deeper wallet, budget and x402 mechanics — including calling the HTTP API dire | Balance, funding, spend caps | `blockrun_wallet` | | The crypto tools overlap heavily. Prefer the free ones (`blockrun_price`, `blockrun_dex`) when -they already answer the question, and reach for `blockrun_surf` only when they do not. +they already answer the question, and reach for `blockrun_defi` or `blockrun_markets` only when +they do not. On-chain SQL, wallet labels and social mindshare came from `blockrun_surf`, which +was removed in 0.49.0 — the gateway retired Surf on 2026-09-06 and there is no replacement yet. ## Picking a chat model diff --git a/skills/crypto-data/SKILL.md b/skills/crypto-data/SKILL.md index 7d85cf4..df69729 100644 --- a/skills/crypto-data/SKILL.md +++ b/skills/crypto-data/SKILL.md @@ -1,6 +1,6 @@ --- name: crypto-data -description: Use for any crypto data question — token/coin prices, FX, commodities, stocks, OHLC history, DEX pairs and liquidity, DeFi TVL, yield/APY pools, on-chain SQL, wallet labels and net worth, social mindshare, news, or raw JSON-RPC against a chain. Routes across five tools that overlap heavily, so it also says which one to use and which are FREE — blockrun_price (crypto/FX/commodities free, Pyth), blockrun_dex (free, DexScreener), blockrun_defi (DefiLlama TVL + yields), blockrun_surf (83 endpoints — on-chain SQL, 100M+ wallet labels, social), blockrun_rpc (40 chains). No API keys, pay-per-call in USDC via x402. +description: "Use for any crypto data question — token/coin prices, FX, commodities, stocks, OHLC history, DEX pairs and liquidity, DeFi TVL, yield/APY pools, or raw JSON-RPC against a chain; also when the user asks for on-chain SQL, wallet labels/net worth, social mindshare or crypto news, so they are told plainly what BlockRun serves and what it does not. Routes across four live tools that overlap and says which one to use and which are FREE — blockrun_price (crypto/FX/commodities free, Pyth), blockrun_dex (free, DexScreener), blockrun_defi (DefiLlama TVL + yields), blockrun_rpc (40 chains). blockrun_surf is retired (gateway 410 since 2026-09-06) and must not be called for data. No API keys, pay-per-call in USDC via x402." triggers: - "crypto price" - "token price" @@ -49,7 +49,9 @@ triggers: # Crypto Data -Five tools cover crypto data and they overlap. **Pick by cost first** — two of them are free, and paying for a quote you could get for nothing is the most common mistake here. +Four live tools cover crypto data and they overlap. **Pick by cost first** — two of them are free, and paying for a quote you could get for nothing is the most common mistake here. + +A fifth tool, `blockrun_surf`, was **removed in 0.49.0**: the gateway has answered every Surf path with HTTP 410 since 2026-09-06. Do not name it; the [`surf` skill](../surf/SKILL.md) maps each former Surf capability to where it lives now — and lists the ones (on-chain SQL, cross-chain wallet labels, social mindshare, CEX order books) that have no BlockRun source yet. ## Route by cost — check this before calling anything @@ -60,22 +62,22 @@ Five tools cover crypto data and they overlap. **Pick by cost first** — two of | **FX rate / commodity (gold, oil)** | `blockrun_price` category:"fx" or `"commodity"` | **FREE** | | **Which symbols exist?** | `blockrun_price` action:"list" | **FREE** | | **DEX pair, liquidity, volume, contract** | `blockrun_dex` | **FREE** | -| Stock quote / history (12 markets) | `blockrun_price` category:"stocks" | $0.0020 | +| Stock ticker catalog (12 markets) | `blockrun_price` action:"list" category:"stocks" | **FREE** — quotes/history withdrawn 2026-09-05 (501) | | Token price by contract address | `blockrun_defi` path:"prices/{coins}" | $0.0020 | | Raw JSON-RPC on 40 chains | `blockrun_rpc` | $0.0030 | | Protocol TVL, chain TVL, yields/APY | `blockrun_defi` | $0.0060 | -| **Everything below** (on-chain SQL, wallet labels, social, news, unlocks, liquidations, ETF flows) | `blockrun_surf` | $0.0085 | -| Raw SQL over 80+ ClickHouse tables | `blockrun_surf` path:"onchain/sql" | $0.0085 | +| Who a Polymarket wallet is, and which wallets are linked to it | `blockrun_markets` `polymarket/wallet/identity/{w}`, `.../cluster` | $0.0085 | +| On-chain SQL, cross-chain wallet labels / net worth, social mindshare, CEX order books, ETF flows, unlocks | **nothing on BlockRun yet** — Surf retired 2026-09-06; say so | — | -Every price below is what x402 actually **charges** (the base plus the gateway's $0.001 flat fee), verified against live `payment-required` headers — not the base you may see in a 402 body. +Every price below is what x402 actually **charges on Base** (the base plus the gateway's $0.001 flat fee), verified against live `payment-required` headers — not the base you may see in a 402 body. The Solana gateway quotes the base alone; the account rail charges no fee. -**The rule:** a plain crypto price or a DEX pair is free. Only reach for `blockrun_surf` when you need something the free tools genuinely do not have — labels, SQL, social, news, unlocks. +**The rule:** a plain crypto price or a DEX pair is free. When the question needs something the four live tools do not have — labels, SQL, social, news, unlocks — tell the user BlockRun does not serve it right now. The tool that used to (`blockrun_surf`) was removed in 0.49.0. -**Prediction markets are never a Surf question.** Surf carries `prediction-market/*` endpoints, but Predexon (`blockrun_markets`) serves the same Polymarket/Kalshi data at the **same $0.0085** — and adds wallet clustering, smart money, sports, UMA, and five more venues that Surf does not have at all. Price is no longer the argument (it was 7.5× cheaper before 2026-07-15); coverage is, and it is decisive. Route odds, positions and market history to [`skills/prediction-markets/SKILL.md`](../prediction-markets/SKILL.md). +**Prediction markets go to `blockrun_markets`** (Predexon): Polymarket, Kalshi, Limitless, Opinion and Predict.Fun, plus wallet clustering, smart money and UMA. For sports odds use `markets/search` with `{ q: "NBA" }` or `polymarket/events` with `{ search: "NBA" }` — the dedicated `sports/*` routes are degraded upstream, and the bare `markets` route with a `league` filter was removed on 2026-08-04 and 404s. Route odds, positions and market history to [`skills/prediction-markets/SKILL.md`](../prediction-markets/SKILL.md). ## blockrun_price — quotes & history (Pyth-backed) -Free for crypto, FX and commodities. $0.0020 only for stocks. +Free for crypto, FX and commodities. Equity is catalog-only: since 2026-09-05 the gateway answers `stocks` price/history with a pre-payment 501 ("We do not currently serve equity prices") — nothing is charged, retrying does not help, and the user should contact hello@blockrun.ai for equity coverage. `action:"list"` still returns the ticker catalog for free. ```ts blockrun_price({ action: "price", category: "crypto", symbol: "BTC-USD" }) // FREE @@ -84,10 +86,10 @@ blockrun_price({ action: "history", category: "crypto", symbol: "ETH-USD", blockrun_price({ action: "price", category: "fx", symbol: "EUR-USD" }) // FREE blockrun_price({ action: "price", category: "commodity", symbol: "XAU-USD" }) // FREE — gold blockrun_price({ action: "list", category: "crypto" }) // FREE — discovery -blockrun_price({ action: "price", category: "stocks", symbol: "AAPL", market: "us" }) // $0.0020 +blockrun_price({ action: "list", category: "stocks", market: "us", query: "AAPL" }) // FREE — catalog only; price/history → 501 ``` -Stock markets: `us`, `hk`, `jp`, `kr`, `gb`, `de`, `fr`, `nl`, `ie`, `lu`, `cn`, `ca` — `market` is required when `category:"stocks"`. +Stock markets: `us`, `hk`, `jp`, `kr`, `gb`, `de`, `fr`, `nl`, `ie`, `lu`, `cn`, `ca` — `market` is required when `category:"stocks"`. Do not call `action:"price"` or `"history"` on `stocks`: the gateway does not serve equity quotes right now. ## blockrun_dex — DEX pairs & liquidity (DexScreener) @@ -122,20 +124,9 @@ blockrun_defi({ path: "yields" }) // big payload — filter af blockrun_defi({ path: "prices/coingecko:ethereum" }) ``` -## blockrun_surf — the things nothing else has - -83 endpoints. Reach here when the free tools cannot answer it: **on-chain SQL, 100M+ labeled wallets across 13 chains, social/CT intelligence, news, tokenomics, liquidations, ETF flows, VC portfolios.** Full catalog and recipes: [`skills/surf/SKILL.md`](../surf/SKILL.md). +## blockrun_surf — removed 2026-09-06 (tool dropped in 0.49.0) -```ts -blockrun_surf({ path: "wallet/labels/batch", params: { addresses: "0xabc,0xdef" } }) // CEX/Whale/MEV/Bot -blockrun_surf({ path: "wallet/net-worth", params: { address: "0xabc" } }) -blockrun_surf({ path: "token/tokenomics", params: { symbol: "ARB" } }) // unlocks + vesting -blockrun_surf({ path: "market/etf", params: { symbol: "BTC" } }) // ETF flows -blockrun_surf({ path: "social/mindshare", params: { project: "base" } }) -blockrun_surf({ path: "onchain/sql", body: { - sql: "SELECT token_address, count() FROM ethereum.dex_trades WHERE block_time > now() - INTERVAL 1 DAY GROUP BY 1 ORDER BY 2 DESC LIMIT 10" -}}) // $0.0085 -``` +The gateway answers every `/v1/surf/*` path with `410 endpoint_retired` (verified live 2026-09-08; `sol.blockrun.ai` 404s; `/api/openapi` lists no Surf route), so the tool was removed from the server rather than left to return an error. It is not one of the 19 tools. Route former Surf questions to `blockrun_price`, `blockrun_dex`, `blockrun_defi`, `blockrun_markets` or `blockrun_rpc`; see [`skills/surf/SKILL.md`](../surf/SKILL.md) for the endpoint-by-endpoint map and for what has no replacement. ## blockrun_rpc — raw chain access @@ -153,26 +144,32 @@ blockrun_rpc({ network: "base", method: "eth_blockNumber", params: [] }) blockrun_price({ action: "price", category: "crypto", symbol: "BTC-USD" }) // FREE ``` -Not `blockrun_surf({ path: "market/price" })` — that is $0.0085 for an answer you can get free. +Not `blockrun_surf` — it no longer exists (removed in 0.49.0); before 2026-09-06 it charged $0.0085 for an answer you can get free. ### 2. "Is this token legit?" ← compound, mostly free ```ts -blockrun_dex({ token: "0xCONTRACT" }) // FREE — liquidity, volume, pairs -blockrun_surf({ path: "token/holders", params: { address: "0xCONTRACT" } }) // concentration -blockrun_surf({ path: "token/tokenomics", params: { symbol: "TKN" } }) // unlock cliff coming? +blockrun_dex({ token: "0xCONTRACT" }) // FREE — liquidity, volume, pairs, age +blockrun_defi({ path: "prices/base:0xCONTRACT" }) // $0.0020 — is it priced by DefiLlama at all? +blockrun_rpc({ network: "base", method: "eth_call", params: [{ to: "0xCONTRACT", data: "0x18160ddd" }, "latest"] }) // totalSupply() ``` -Start free. Only pay once the free signal says it is worth a closer look. +Start free. Only pay once the free signal says it is worth a closer look. Holder concentration and unlock schedules were Surf features and have no BlockRun source right now — say so rather than guessing. ### 3. "Who owns this wallet and what do they hold?" ```ts -blockrun_surf({ path: "wallet/labels/batch", params: { addresses: "0xWHALE" } }) // CEX? Whale? MEV bot? -blockrun_surf({ path: "wallet/net-worth", params: { address: "0xWHALE" } }) -blockrun_surf({ path: "wallet/protocols", params: { address: "0xWHALE" } }) // Aave/Lido/Uni positions +// A Polymarket trader: identity, linked wallets, P&L — via Predexon +blockrun_markets({ path: "polymarket/wallet/identity/0xWHALE" }) +blockrun_markets({ path: "polymarket/wallet/0xWHALE/cluster" }) +blockrun_markets({ path: "polymarket/wallet/pnl/0xWHALE" }) + +// Any address: balances via raw RPC +blockrun_rpc({ network: "ethereum", method: "eth_getBalance", params: ["0xWHALE", "latest"] }) ``` +Cross-chain labels (CEX / MEV / bridge), net-worth history and DeFi position breakdowns were Surf features; nothing on BlockRun serves them today. + ### 4. "Where's the best yield right now?" ```ts @@ -183,12 +180,15 @@ blockrun_defi({ path: "chains" }) // where the money actually is ### 5. "Give me the macro picture" ```ts -blockrun_surf({ path: "market/etf", params: { symbol: "BTC" } }) -blockrun_surf({ path: "exchange/funding-history", params: { symbol: "BTCUSDT" } }) -blockrun_surf({ path: "market/liquidation/chart" }) -blockrun_surf({ path: "market/fear-greed" }) +blockrun_price({ action: "price", category: "crypto", symbol: "BTC-USD" }) // FREE +blockrun_price({ action: "price", category: "commodity", symbol: "XAU-USD" }) // FREE — gold +blockrun_price({ action: "price", category: "fx", symbol: "EUR-USD" }) // FREE +blockrun_defi({ path: "chains" }) // $0.0060 — where DeFi capital sits +blockrun_markets({ path: "markets/search", params: { q: "bitcoin", status: "open" } }) // $0.0085 — what the crowd is pricing ``` +ETF flows, funding rates, liquidation charts and the Fear & Greed index were Surf features with no BlockRun source today. + ### 6. "Is gold up today?" — also free ```ts @@ -199,5 +199,5 @@ blockrun_price({ action: "price", category: "commodity", symbol: "XAU-USD" }) - **Prediction markets** (odds, smart money, wallet clustering) → [`skills/prediction-markets/SKILL.md`](../prediction-markets/SKILL.md) - **Polymarket trading** (real money) → [`skills/polymarket-trading/SKILL.md`](../polymarket-trading/SKILL.md) -- **All 83 Surf endpoints** → [`skills/surf/SKILL.md`](../surf/SKILL.md) +- **Surf is retired** — what replaced each former endpoint → [`skills/surf/SKILL.md`](../surf/SKILL.md) - **Raw RPC** → [`skills/rpc/SKILL.md`](../rpc/SKILL.md) diff --git a/skills/gentech-blockrun/SKILL.md b/skills/gentech-blockrun/SKILL.md index 6dc239c..11c5247 100644 --- a/skills/gentech-blockrun/SKILL.md +++ b/skills/gentech-blockrun/SKILL.md @@ -66,8 +66,8 @@ blockrun_wallet(action="status") ### Pattern 1: Regular Price Checks (FREE) Crypto, FX and commodity quotes cost **nothing** — `blockrun_price` is free for those -categories (only `stocks`/`usstock` is paid, at $0.0020). Use them liberally, and do -not pay $0.0085 to `blockrun_surf` for a quote you can get for $0. +categories (`stocks`/`usstock` quotes are not served since 2026-09-05 — the gateway 501s before payment; only the ticker catalog works). Use them liberally. +(`blockrun_surf`, which used to charge $0.0085 for the same quote, was removed in 0.49.0 — the gateway 410s every Surf path since 2026-09-06.) ```python # Single price @@ -81,7 +81,7 @@ blockrun_price(action="price", category="crypto", symbol="SOL-USD") blockrun_price(action="list", category="crypto", query="sol") ``` -**Cost:** $0 — crypto/FX/commodity price *and* list calls are both free. Only `category:"stocks"` is paid ($0.0020). +**Cost:** $0 — crypto/FX/commodity price *and* list calls are both free. `category:"stocks"` price/history currently return 501 (equity quotes withdrawn 2026-09-05, nothing charged); its `list` catalog is free. ### Pattern 2: Token Research Pipeline (~$0.018) @@ -132,7 +132,7 @@ blockrun_wallet(action="delegate", agent_id="research", agent_limit=2.0) blockrun_wallet(action="delegate", agent_id="content", agent_limit=1.0) # Pass agent_id on every call -blockrun_surf({ path: "market/price", params: { symbol: "BTC" }, agent_id: "research" }) +blockrun_defi({ path: "protocols", agent_id: "research" }) blockrun_search({ body: { query: "latest news", sources: ["web"] }, agent_id: "research" }) # Audit at end of day @@ -204,12 +204,11 @@ blockrun_video({ prompt: "animated data visualization", duration_seconds: 8 }) | `blockrun_dex` | FREE | unlimited | | `blockrun_rpc` | $0.0030 | on-chain reads (batch: $0.002/element + $0.001) | | `blockrun_wallet` (status/report) | FREE | before every session | -| `blockrun_price` (quote) | **FREE** | crypto/FX/commodity; stocks $0.0020 | +| `blockrun_price` (quote) | **FREE** | crypto/FX/commodity; stocks quotes not served (501 since 2026-09-05) | | `blockrun_defi` | $0.0060 | protocol/chain analysis ($0.0020 for prices/*) | | `blockrun_chat` (free mode) | $0 | NVIDIA-backed chat | | `blockrun_chat` (glm mode) | per-token | Zhipu GLM-5 coding — billed on tokens used, not a flat rate | | `blockrun_exa` (search) | $0.0110 | deep research (`contents`: $0.002/URL + $0.001) | -| `blockrun_surf` | $0.0085 | crypto data, wallets/candles/search, on-chain SQL (flat) | | `blockrun_speech` | $0.0535/1k chars | TTS | | `blockrun_image` | $0.01675–0.106 | image generation | | `blockrun_music` | $0.1585 | music tracks | @@ -224,7 +223,7 @@ blockrun_video({ prompt: "animated data visualization", duration_seconds: 8 }) # ~$0.30/day typical Daily budget: Free tools: unlimited (price quotes, dex, list, wallet status) - Crypto data: ~$0.10 (surf + markets, flat $0.0085 each; defi $0.0060) + Crypto data: ~$0.10 (markets, flat $0.0085; defi $0.0060; surf is retired and free) AI calls: ~$0.10 (chat free mode, occasional glm) Media: ~$0.10 (occasional image/speech) Total: ~$0.30/day @@ -292,7 +291,7 @@ blockrun_wallet(action="chain", chain="base") blockrun_wallet(action="chain") ``` -**Note:** Base is required for music, speech, and realface. Image and video pay on either chain. Solana works for price, wallet, dex, rpc, surf, etc. +**Note:** Base is required for music, speech, and realface. Image and video pay on either chain. Solana works for price, wallet, dex, rpc, markets, etc. --- diff --git a/skills/phone/SKILL.md b/skills/phone/SKILL.md index d37c909..c756553 100644 --- a/skills/phone/SKILL.md +++ b/skills/phone/SKILL.md @@ -20,7 +20,7 @@ triggers: # Phone & Voice -Two namespaces in one tool: **`/v1/phone/*`** for number intelligence + provisioning, **`/v1/voice/*`** for outbound AI calls. Pay per call in USDC. +Two namespaces in one tool: **`/v1/phone/*`** for number intelligence + provisioning, **`/v1/voice/*`** for outbound AI calls. Pay per call in USDC. Any other path is refused before payment — the tool never reaches routes outside these two. Phone numbers use **E.164 format** — `+` followed by country code and subscriber digits (US: `+1` + 10 digits; UK: `+44` + 10 digits; etc.). The examples below use `<+E.164-number>` as a placeholder — the LLM should substitute the actual number from the user's request, not copy the literal placeholder. diff --git a/skills/polymarket-trading/SKILL.md b/skills/polymarket-trading/SKILL.md index 7f9b3ec..b5ac721 100644 --- a/skills/polymarket-trading/SKILL.md +++ b/skills/polymarket-trading/SKILL.md @@ -32,7 +32,10 @@ this tool only trades. trade.** Call once WITHOUT confirm → show the dry-run preview → ask → re-call with `confirm:true`. 2. Per-order cap `POLYMARKET_MAX_BET_USD` (default $25) and optional session cap - are enforced server-side; don't try to split orders to sneak past them. + are enforced server-side; don't try to split orders to sneak past them. The + optional `POLYMARKET_MAX_FUND_USD` bounds a single `fund` call the same way. + A `withdraw` preview marks any `to_address` that is not the user's own agent + wallet as `CUSTOM` — show the user that line verbatim before confirming. 3. On ANY error, read the message — it says exactly what to do next (fund, approve, region, re-run setup). Don't retry blindly. @@ -84,6 +87,8 @@ blockrun_polymarket action:"withdraw" confirm:true # (partial: ## Order semantics - Prices are probabilities 0–1, auto-rounded to the market's tick grid. +- A market-order preview states `worst fill ≤ X` (buy) / `≥ X` (sell); the + order is signed at that bound, so quote it to the user as the price limit. - Market **buy** = `amount_usd` (dollars). Market **sell** = `size` (shares). - Limit orders: `price` + `size`; default GTC; `post_only:true` for maker-only. - FOK fails whole-or-nothing; FAK fills what it can. On "FOK not filled", offer diff --git a/skills/prediction-markets/SKILL.md b/skills/prediction-markets/SKILL.md index 4d3015d..bbc8bc5 100644 --- a/skills/prediction-markets/SKILL.md +++ b/skills/prediction-markets/SKILL.md @@ -96,13 +96,16 @@ Current parameter contracts that prevent paid 4xx responses: `min_profit_factor`. `window` only scopes the time range and is **not** sufficient alone (verified: window-only returns a paid 400). Use `{ window: "30d", min_trades: "100" }`; narrower cohorts are fine. -- `markets/listings` is retired upstream (410 Gone) — the MCP blocks it before payment. +- `markets/listings` is retired upstream (410 Gone) — the MCP blocks it before payment. The rest of the + canonical layer went with it on 2026-08-04: bare `markets`, `outcomes/{predexon_id}`, `matching-markets` + and `matching-markets/pairs` all return **404 Unknown Predexon endpoint before payment** (verified live + 2026-09-08). Only `markets/search` survived. No live `/v1/pm` route accepts a `league` param. ## Two Pricing Tiers | Tier | Price | What | |---|---|---| -| **All endpoints** | $0.0085 | Market data, events, history, candles, orderbooks, trades, leaderboard, sports, UMA, wallet analytics, smart money, identity + clustering, cross-venue matching, Binance | +| **All endpoints** | $0.0085 on Base ($0.0075 + the $0.001 network fee; Solana quotes $0.0075) | Market data, events, history, candles, orderbooks, trades, leaderboard, UMA, wallet analytics, smart money, identity + clustering, cross-venue search, Binance (`sports/*` is degraded upstream — see below; the gateway releases the payment on the upstream 500) | Pass-through pricing, 0% BlockRun margin — settles straight to Predexon's Base treasury. @@ -110,11 +113,7 @@ Pass-through pricing, 0% BlockRun margin — settles straight to Predexon's Base | User wants… | path | Tier | |---|---|---| -| **Same question across venues** | `markets` | 1 | -| **Search every venue at once** | `markets/search` | 2 | -| Resolve a canonical outcome ID | `outcomes/{predexon_id}` | 1 | -| **Equivalent markets (arbitrage)** | `matching-markets` | 2 | -| Active matched pairs | `matching-markets/pairs` | 2 | +| **Same question across venues / search every venue at once** | `markets/search` (`q`) | 2 | | Active Polymarket events | `polymarket/events` | 1 | | Polymarket markets | `polymarket/markets` | 1 | | Large result sets (stable paging) | `polymarket/markets/keyset` | 1 | @@ -150,10 +149,10 @@ Pass-through pricing, 0% BlockRun margin — settles straight to Predexon's Base | UMA status + timeline for a market | `polymarket/uma/market/{condition_id}` | 1 | | Kalshi markets | `kalshi/markets` | 1 | | Kalshi trades / orderbooks | `kalshi/trades`, `kalshi/orderbooks` | 1 | -| Sports categories | `sports/categories` | 1 | -| Sports markets by league | `sports/markets` | 1 | -| One game, all venue outcomes | `sports/markets/{game_id}` | 1 | -| Equivalent sports outcomes | `sports/outcomes/{predexon_id}` | 1 | +| ⚠ Sports categories — **degraded upstream since 2026-08-04, do not call** | `sports/categories` | 1 | +| ⚠ Sports markets by league — degraded, use `markets/search` + `q=` or `polymarket/events` + `search=` | `sports/markets` | 1 | +| ⚠ One game, all venue outcomes — degraded | `sports/markets/{game_id}` | 1 | +| ⚠ Equivalent sports outcomes — degraded | `sports/outcomes/{predexon_id}` | 1 | | Limitless / Opinion / Predict.Fun markets | `limitless/markets`, `opinion/markets`, `predictfun/markets` | 1 | | Their historical orderbook snapshots | `limitless/orderbooks`, `opinion/orderbooks`, `predictfun/orderbooks` | 1 | | Binance candles / ticks | `binance/candles/{symbol}`, `binance/ticks/{symbol}` | 2 | @@ -168,11 +167,11 @@ blockrun_markets({ path: "polymarket/events", params: { limit: "10" } }) ### 2. "What's the market saying about the 2028 election?" -Search every venue in one call, then resolve the canonical outcome. +Search every venue in one call — each hit carries its venue and IDs — then pull the chosen Polymarket market by `condition_id`. ```ts -blockrun_markets({ path: "markets/search", params: { q: "2028 presidential election" } }) -blockrun_markets({ path: "outcomes/PXM-12345" }) // → venue listings + prices side by side +blockrun_markets({ path: "markets/search", params: { q: "2028 presidential election", status: "open" } }) +blockrun_markets({ path: "polymarket/markets/keyset", params: { condition_id: "0xCONDITION_ID" } }) // full market record ``` ### 3. "Show me this market's price history" (impossible from a free API) @@ -228,18 +227,29 @@ Then trade it with `blockrun_polymarket` (see `skills/polymarket-trading/SKILL.m ### 7. "Is the same bet cheaper on another venue?" ← arbitrage +`matching-markets` and `matching-markets/pairs` were removed upstream (404 before payment). One search returns +the same question from every venue; compare the prices in the result. + ```ts -blockrun_markets({ path: "matching-markets", params: { status: "active" } }) -blockrun_markets({ path: "matching-markets/pairs" }) +blockrun_markets({ path: "markets/search", params: { q: "Fed cuts rates in December", status: "open" } }) +// → Polymarket, Kalshi, Limitless, Opinion, Predict.Fun hits side by side; the spread is the arbitrage ``` ### 8. "Who's ahead in tonight's NBA games?" ```ts -blockrun_markets({ path: "sports/markets", params: { league: "NBA", status: "open" } }) -blockrun_markets({ path: "sports/markets/GAME_ID" }) // every venue's price for that game +blockrun_markets({ path: "markets/search", params: { q: "NBA", status: "open" } }) // every venue's NBA markets +blockrun_markets({ path: "polymarket/events", params: { search: "NBA", status: "open" } }) // Polymarket game events +blockrun_markets({ path: "kalshi/markets", params: { search: "NBA" } }) // Kalshi's ``` +Do **not** route this to `sports/*`. All four `sports/*` paths have returned a Predexon 500 on every call since +2026-08-04 (re-verified 2026-09-08). The gateway still routes them but withdrew them from discovery, and it +releases the payment on that upstream 500 — the tool says "nothing was charged" when the gateway's own +"payment NOT charged" confirmation is in the response, and otherwise points at `blockrun_wallet action:"report"`. +Do not use `markets` with `league=` either: that route was removed on 2026-08-04 and 404s before payment, and no +live `/v1/pm` route accepts `league`. The sports routes come back here the day Predexon repairs them. + ### 9. "Is this market about to resolve?" ```ts diff --git a/skills/rpc/SKILL.md b/skills/rpc/SKILL.md index a3ba7dd..f494803 100644 --- a/skills/rpc/SKILL.md +++ b/skills/rpc/SKILL.md @@ -26,7 +26,7 @@ triggers: |---|---| | "What's ETH trading at?" | `blockrun_price` (free) | | "PEPE/WETH pool liquidity?" | `blockrun_dex` (free) | -| "What's this wallet labeled as / holding?" | `blockrun_surf` | +| "What's this wallet holding?" | `blockrun_rpc` (labels: no tool since Surf was retired 2026-09-06) | | "Call `balanceOf(0x...)` on this ERC-20" | **`blockrun_rpc`** | | "Latest block / tx receipt / event logs / gas price" | **`blockrun_rpc`** | | "Solana account info / slot / signatures" | **`blockrun_rpc`** | diff --git a/skills/signal-to-trade-demo/references/demo-cases.md b/skills/signal-to-trade-demo/references/demo-cases.md index 8cf4dea..e47e423 100644 --- a/skills/signal-to-trade-demo/references/demo-cases.md +++ b/skills/signal-to-trade-demo/references/demo-cases.md @@ -17,7 +17,7 @@ volume is zero, no Yes/No token is present, or the dry-run finds no book. Search an open Fed/rates or inflation question with a precise resolution rule. Use the same market-history, smart-money, and book lenses. Add live news only -when the Trading profile's `blockrun_surf` can cite a current source. Keep news +when `blockrun_search` or `blockrun_exa` can cite a current source. Keep news and market-implied probability separate. ## Case C — crypto up/down (short-form fallback) diff --git a/skills/surf/SKILL.md b/skills/surf/SKILL.md index 34df649..285de5e 100644 --- a/skills/surf/SKILL.md +++ b/skills/surf/SKILL.md @@ -1,6 +1,6 @@ --- name: surf -description: Use when the user wants crypto data — token prices, on-chain SQL, prediction-market positions, CEX order books, wallet labels/net-worth, social mindshare, news, or unified search. 83 endpoints across exchange, on-chain, wallet, social, prediction, news and search — one API, flat $0.0085/call in USDC via x402. Settles directly to Surf's Base treasury; no Surf account needed. +description: "Surf (asksurf.ai) is RETIRED on BlockRun and the blockrun_surf tool was REMOVED in 0.49.0 — the gateway has answered every /v1/surf/* path with HTTP 410 endpoint_retired since 2026-09-06. Use this skill to route a former Surf question (on-chain SQL, wallet labels and net worth, CEX order books, social mindshare, news) to the tool that still serves it, and to say plainly what has no replacement yet." triggers: - "surf" - "asksurf" @@ -31,254 +31,87 @@ triggers: - "kalshi data" --- -# Surf — Crypto Data via BlockRun +# Surf — retired on BlockRun (2026-09-06) -Surf (asksurf.ai) aggregates **83 crypto data endpoints** across CEX market data, on-chain SQL (13 chains, 80+ ClickHouse tables), 100M+ labeled wallets, prediction markets (Polymarket + Kalshi side-by-side), social/CT intelligence, news and unified search. +Surf (asksurf.ai) is no longer served through BlockRun. Since **2026-09-06** the +gateway answers every `/v1/surf/*` path with: -BlockRun is Surf's x402 payment rail — every call settles **directly to Surf's Base treasury**. You hold the wallet, BlockRun holds the Surf key, Surf holds the data. No Surf account, no API key, no monthly minimum. - -## How to Call from MCP - -One tool, three params. The MCP tool auto-routes method (POST when `body` is set, GET otherwise) and auto-validates required params before settling: - -```ts -blockrun_surf({ path: "market/price", params: { symbol: "BTC" } }) - -blockrun_surf({ path: "onchain/sql", body: { - sql: "SELECT token_address, count() FROM ethereum.dex_trades WHERE block_time > now() - INTERVAL 1 DAY GROUP BY 1 ORDER BY 2 DESC LIMIT 10" -}}) - -blockrun_surf({ path: "wallet/labels/batch", params: { addresses: "0xabc,0xdef" } }) -``` - -## Pricing — one flat rate - -**$0.0085 per call. Every endpoint, no tiers, including raw on-chain SQL.** - -Verified against the gateway's own `payment-required` header, which is free to request — send any call with no payment header and it quotes the exact charge. All of `market/price`, `wallet/labels/batch`, `social/mindshare`, `news/feed`, `exchange/klines`, `search/web` **and `onchain/sql`** return the same $0.0085 (`SURF_TIER_1/2/3_PRICE` are all identical upstream). - -Ignore any "premium tier" pricing you may have seen — SQL used to cost more and no longer does. - -Wrong / missing required params return HTTP 400 **without charging** — pre-validation runs before settlement. - -## Do NOT use Surf for prediction markets - -Surf carries 17 `prediction-market/*` endpoints (Polymarket + Kalshi). **Use `blockrun_markets` (Predexon) instead — same data, same price, and far deeper.** - -Predexon used to be 7.5× cheaper; since 2026-07-15 both bill the same flat rate, so the choice is now purely about coverage — and Predexon still wins on coverage by a wide margin. - -| | Predexon (`blockrun_markets`) | Surf | -|---|---|---| -| Polymarket / Kalshi markets | $0.0085 | $0.0085 (same) | -| Wallet clustering, smart money, leaderboards | ✅ | ❌ none | -| Limitless, Opinion, Predict.Fun, sports, UMA | ✅ | ❌ Polymarket + Kalshi only | - -The only Surf prediction-market endpoint with no Predexon equivalent is `prediction-market/category-metrics`. Everything else is a strictly worse buy. See [`skills/prediction-markets/SKILL.md`](../prediction-markets/SKILL.md). - -**Reach for Surf when Predexon cannot answer it:** on-chain SQL, 100M+ wallet labels across 13 chains, 16 CEXs, social/CT intelligence, news, tokenomics/unlocks, liquidations, ETF flows, VC portfolios. Predexon has none of those. - -## Quick Decision Table — "User asks about X" - -| User wants… | Method | Path | Required | -|---|---|---|---| -| BTC/ETH price | GET | `market/price` | `symbol` | -| ETF flow history | GET | `market/etf` | `symbol` | -| Fear & Greed index | GET | `market/fear-greed` | – | -| Top 100 tokens by market cap | GET | `market/ranking` | – | -| Options skew / IV / volume | GET | `market/options` | `symbol` | -| CEX ticker for a pair | GET | `exchange/price` | `pair` | -| Perp snapshot (funding + OI) | GET | `exchange/perp` | `pair` | -| Order book depth | GET | `exchange/depth` | `pair` | -| OHLCV candles | GET | `exchange/klines` | `pair` | -| Funding rate history | GET | `exchange/funding-history` | `pair` | -| Long/short ratio | GET | `exchange/long-short-ratio` | `pair` | -| Bridge protocols by volume | GET | `onchain/bridge/ranking` | – | -| Yield pool ranking | GET | `onchain/yield/ranking` | – | -| Current gas price (per chain) | GET | `onchain/gas-price` | `chain` | -| Transaction details | GET | `onchain/tx` | `hash`, `chain` | -| **Raw on-chain SQL** | POST | `onchain/sql` | body: `sql` | -| **Structured on-chain query** | POST | `onchain/query` | body: typed predicates | -| Inspect ClickHouse schema | GET | `onchain/schema` | – | -| Polymarket markets ranking | GET | `prediction-market/polymarket/ranking` | – | -| Polymarket price history | GET | `prediction-market/polymarket/prices` | `condition_id` | -| Polymarket positions for wallet | GET | `prediction-market/polymarket/positions` | `address` | -| Kalshi markets ranking | GET | `prediction-market/kalshi/ranking` | – | -| Kalshi market detail | GET | `prediction-market/kalshi/markets` | `market_ticker` | -| **Search Polymarket / Kalshi** | GET | `search/polymarket` / `search/kalshi` | – | -| Wallet profile (cross-chain) | GET | `wallet/detail` | `address` | -| Wallet net-worth time series | GET | `wallet/net-worth` | `address` | -| Wallet DeFi positions | GET | `wallet/protocols` | `address` | -| **Batch wallet labels (CEX/Whale/MEV…)** | GET | `wallet/labels/batch` | `addresses` | -| Token tokenomics + unlocks | GET | `token/tokenomics` | – | -| Token holders top N | GET | `token/holders` | `address`, `chain` | -| Token transfers | GET | `token/transfers` | `address`, `chain` | -| Token DEX trades | GET | `token/dex-trades` | `address` | -| Social mindshare time series | GET | `social/mindshare` | `q`, `interval` | -| Smart-follower history | GET | `social/smart-followers/history` | – | -| Twitter user profile | GET | `social/user` | `handle` | -| Twitter user posts | GET | `social/user/posts` | `handle` | -| Tweet replies | GET | `social/tweet/replies` | `tweet_id` | -| **Web search (crypto-scoped)** | GET | `search/web` | `q` | -| News article search | GET | `search/news` | `q` | -| KOL / CT people search | GET | `search/social/people` | `q` | -| Tweet full-text search | GET | `search/social/posts` | `q` | -| Project / token search | GET | `search/project` | `q` | -| Wallet search (by ENS, label) | GET | `search/wallet` | `q` | -| VC fund portfolio | GET | `fund/portfolio` | – | -| VC fund ranking | GET | `fund/ranking` | `metric` | -| DeFi protocol ranking | GET | `project/defi/ranking` | `metric` | -| Project full profile | GET | `project/detail` | – | -| Clean a webpage to markdown | GET | `web/fetch` | `url` | -| News feed | GET | `news/feed` | – | -| Single news article | GET | `news/detail` | `id` | - -## Worked Examples - -### 1. "What's BTC trading at?" - -```ts -blockrun_surf({ path: "market/price", params: { symbol: "BTC" } }) -``` -**Cost: $0.0085.** Returns price history; latest point = current price. - -### 2. "Top 10 tokens by DEX volume on Ethereum in the last 24h" - -```ts -blockrun_surf({ - path: "onchain/sql", - body: { - sql: ` - SELECT token_address, sum(amount_usd) AS volume_usd - FROM ethereum.dex_trades - WHERE block_time > now() - INTERVAL 1 DAY - GROUP BY token_address - ORDER BY volume_usd DESC - LIMIT 10 - ` - } -}) ``` -**Cost: $0.0085.** Raw ClickHouse — same query language Surf's own UI uses, at the same flat rate as any other Surf read. - -### 3. "Is this whale wallet labeled? What does it hold?" - -```ts -// Step 1 — labels (CEX / Whale / Bridge / MEV / Bot / Fund) -blockrun_surf({ path: "wallet/labels/batch", params: { addresses: "0xabc...,0xdef..." } }) - -// Step 2 — cross-chain holdings + DeFi positions -blockrun_surf({ path: "wallet/detail", params: { address: "0xabc..." } }) -blockrun_surf({ path: "wallet/protocols", params: { address: "0xabc..." } }) - -// Step 3 — net-worth time series -blockrun_surf({ path: "wallet/net-worth", params: { address: "0xabc..." } }) +HTTP 410 +{"error":{"code":"endpoint_retired","message":"The Surf data endpoints are retired. We are looking for a new vendor in this space and expect to publish a replacement under /api/v1/."}, + "retired_on":"2026-09-06", + "alternatives":[{"for":"crypto, equity, FX and commodity prices","endpoint":"/api/v1/crypto/price"}, + {"for":"protocol TVL and yields","endpoint":"/api/v1/defillama/*"}, + {"for":"prediction markets","endpoint":"/api/v1/pm/*"}, + {"for":"DEX quotes and swaps","endpoint":"/api/v1/zerox/*"}]} ``` -**Cost: 4 × $0.0085 = $0.034.** Replaces a Nansen subscription for one-off lookups. - -### 4. "What's the market saying about the 2028 election?" - -```ts -// Compare Polymarket + Kalshi side by side -blockrun_surf({ path: "search/polymarket", params: { q: "2028 US president" } }) -blockrun_surf({ path: "search/kalshi", params: { q: "2028 US president" } }) -// Then pull the order book on the leading market -blockrun_surf({ path: "prediction-market/polymarket/prices", - params: { condition_id: "0x..." } }) -``` +Verified live 2026-09-08 with an unauthenticated GET (a 410 is free to fetch). +`sol.blockrun.ai` answers 404 on the same paths, and `/api/openapi` no longer +lists any Surf route. **No 402 is ever issued, so no payment can be made.** -### 5. "Where's mindshare moving for L1s?" +`blockrun_surf` was **removed from the server in 0.49.0** and is no longer one of +the 19 tools: a tool that can only +return an error is not worth the schema every agent carries on every turn. An +MCP client will answer an unknown-tool error if a config still names it. Nothing +can be charged either way. Do not describe this as a temporary outage — it is +not; route the question with the table below. -```ts -blockrun_surf({ path: "social/mindshare", params: { q: "solana", interval: "1d" } }) -blockrun_surf({ path: "social/mindshare", params: { q: "monad", interval: "1d" } }) -blockrun_surf({ path: "social/ranking" }) -``` +## Where each former Surf capability lives now -### 6. "ETF flows + funding rate + long/short — give me the macro picture" +| The user wants… | Surf used to be | Use now | Cost | +|---|---|---|---| +| BTC/ETH/any coin price, OHLC history | `market/price`, `exchange/klines` | `blockrun_price` action:"price" / "history" category:"crypto" | **FREE** | +| FX, gold, oil | `market/price` | `blockrun_price` category:"fx" / "commodity" | **FREE** | +| Fear & Greed, market ranking by cap | `market/fear-greed`, `market/ranking` | `blockrun_price` action:"list" for the symbol universe; no sentiment index on BlockRun | FREE / none | +| DEX pair, liquidity, volume, token by contract | `token/dex-trades`, `search/project` | `blockrun_dex` | **FREE** | +| Token price by contract address | `market/price` | `blockrun_defi` path:"prices/{coins}" | $0.001 + fee | +| Protocol / chain TVL, yield rankings | `project/defi/ranking`, `onchain/yield/ranking` | `blockrun_defi` path:"protocols" / "chains" / "yields" | $0.005 + fee | +| Polymarket / Kalshi markets, prices, positions | `prediction-market/*`, `search/polymarket`, `search/kalshi` | `blockrun_markets` — see [`skills/prediction-markets/SKILL.md`](../prediction-markets/SKILL.md) | $0.0075 + fee | +| Who is this Polymarket wallet, and which wallets are theirs | `wallet/detail`, `wallet/labels/batch` (for Polymarket traders only) | `blockrun_markets` `polymarket/wallet/identity/{wallet}` + `polymarket/wallet/{address}/cluster` | $0.0075 + fee | +| Gas price, a transaction, a balance, a contract read | `onchain/gas-price`, `onchain/tx` | `blockrun_rpc` (`eth_gasPrice`, `eth_getTransactionByHash`, `eth_getBalance`, `eth_call`) — see [`skills/rpc/SKILL.md`](../rpc/SKILL.md) | $0.002 + fee | +| Web / news search | `search/web`, `search/news`, `news/feed` | `blockrun_exa` (neural) or `blockrun_search` (Grok Live Search, web + X + news) | $0.01 + fee / $0.025 × results | + +"fee" is the gateway's flat network fee — $0.001 per call on Base today; the +Solana gateway quotes the base alone; the account rail charges no fee. + +## What has NO replacement on BlockRun yet + +Say so plainly rather than substituting something that answers a different question: + +- **Raw on-chain SQL** over ClickHouse (`onchain/sql`, `onchain/query`, `onchain/schema`) +- **Wallet labels and net worth across 13 chains** for arbitrary addresses (`wallet/labels/batch`, `wallet/net-worth`, `wallet/protocols`, `wallet/history`). Only Polymarket traders are covered, via `blockrun_markets` identity/cluster above. +- **CEX order books, perp snapshots, funding history, long/short ratio, options skew** (`exchange/*`, `market/options`, `market/futures`) +- **ETF flows, liquidation charts, on-chain indicators** (`market/etf`, `market/liquidation/*`, `market/onchain-indicator`) +- **Social / CT intelligence** — mindshare, smart followers, KOL search, tweet search (`social/*`, `search/social/*`) +- **Tokenomics and unlock schedules, token holders and transfers** (`token/*`) +- **VC fund portfolios and rankings** (`fund/*`) +- **Bridge rankings, airdrop search, project profiles** (`onchain/bridge/ranking`, `search/airdrop`, `project/detail`) +- **Webpage-to-markdown** (`web/fetch`) + +The gateway says a new vendor is pending and that the replacement will be +published under `/api/v1/` and listed at `https://blockrun.ai/api/openapi`. Check +there before promising any of the above. + +## Worked example — what a former Surf request looks like now + +**"Is this whale wallet labeled, and what does it hold?"** (was 4 Surf calls, $0.034) ```ts -blockrun_surf({ path: "market/etf", params: { symbol: "BTC" } }) -blockrun_surf({ path: "market/fear-greed" }) -blockrun_surf({ path: "exchange/funding-history", params: { pair: "BTC-USDT" } }) -blockrun_surf({ path: "exchange/long-short-ratio", params: { pair: "BTC-USDT" } }) -``` -**Cost: 4 × $0.0085 = $0.034.** - -## Method Routing — When to Use `body` - -Pass `body` (POST) only for these three endpoints: - -- `onchain/query` — structured, typed predicates against ClickHouse -- `onchain/sql` — raw SQL string in `{ sql: "..." }` - -Everything else is GET with `params`. - -## Python SDK (for non-MCP use) - -```python -from blockrun_llm import setup_agent_wallet +// If it is a Polymarket trader — identity, linked wallets, P&L: +blockrun_markets({ path: "polymarket/wallet/identity/0xWHALE" }) +blockrun_markets({ path: "polymarket/wallet/0xWHALE/cluster" }) +blockrun_markets({ path: "polymarket/wallet/pnl/0xWHALE" }) -client = setup_agent_wallet() - -# GET — same as blockrun_surf({ path, params }) -price = client._get_with_payment_raw("/v1/surf/market/price", {"symbol": "BTC"}) - -# POST — same as blockrun_surf({ path, body }) -result = client._request_with_payment_raw("/v1/surf/onchain/sql", { - "sql": "SELECT count() FROM ethereum.transactions WHERE block_time > now() - INTERVAL 1 HOUR" -}) +// For any address — current native + token balances via raw RPC (free tools first): +blockrun_rpc({ network: "ethereum", method: "eth_getBalance", params: ["0xWHALE", "latest"] }) ``` -## Full Endpoint Catalog (83 endpoints, 12 categories) - -### Exchange (CEX) — 7 -`exchange/markets` · `exchange/price` · `exchange/perp` · `exchange/depth` · `exchange/klines` · `exchange/funding-history` · `exchange/long-short-ratio` - -### Fund (VC intelligence) — 3 -`fund/detail` · `fund/portfolio` · `fund/ranking` - -### Market — 11 -`market/ranking` · `market/fear-greed` · `market/futures` · `market/price` · `market/etf` · `market/options` · `market/liquidation/exchange-list` · `market/liquidation/order` · `market/liquidation/chart` · `market/onchain-indicator` · `market/price-indicator` - -### News — 2 -`news/feed` · `news/detail` - -### On-chain — 7 -`onchain/bridge/ranking` · `onchain/yield/ranking` · `onchain/gas-price` · `onchain/tx` · `onchain/schema` · `onchain/query` (POST) · `onchain/sql` (POST) - -### Prediction Markets — 17 -**Polymarket**: `prediction-market/polymarket/ranking` · `.../trades` · `.../markets` · `.../events` · `.../prices` · `.../volumes` · `.../open-interest` · `.../positions` · `.../activity` · `prediction-market/category-metrics` -**Kalshi**: `prediction-market/kalshi/ranking` · `.../markets` · `.../events` · `.../prices` · `.../trades` · `.../volumes` · `.../open-interest` - -### Project + DeFi — 3 -`project/detail` · `project/defi/metrics` · `project/defi/ranking` - -### Search — 11 -`search/airdrop` · `search/events` · `search/kalshi` · `search/polymarket` · `search/web` · `search/project` · `search/news` · `search/wallet` · `search/fund` · `search/social/people` · `search/social/posts` - -### Social — 11 -`social/detail` · `social/ranking` · `social/smart-followers/history` · `social/mindshare` · `social/tweets` · `social/tweet/replies` · `social/user` · `social/user/followers` · `social/user/following` · `social/user/posts` · `social/user/replies` - -### Token — 4 -`token/tokenomics` · `token/dex-trades` · `token/holders` · `token/transfers` - -### Wallet — 6 -`wallet/detail` · `wallet/history` · `wallet/net-worth` · `wallet/transfers` · `wallet/protocols` · `wallet/labels/batch` - -### Web — 1 -`web/fetch` - -## Gotchas - -- **Required params:** 56 of 83 endpoints require at least one param. The 402 response and the in-tool route surface which fields are missing. Missing params → 400 + no charge. -- **Solana wallet works too**: `blockrun_surf` routes through whichever chain the BlockRun wallet is on (Base or Solana). Surf settlement always lands in Surf's Base treasury. -- **`onchain/sql` is powerful but unrestricted**: there's no row limit on the server side. Add `LIMIT` to your query or you'll pay for a megabyte of JSON. -- **`X-Payment-Receipt` header** lands on the response with the settlement tx hash — keep it for accounting. +Cross-chain labels (CEX / MEV / bridge) and a net-worth time series are not +available on BlockRun right now — tell the user that. `blockrun_surf` no longer exists. ## Reference -- Surf marketplace page: https://blockrun.ai/marketplace/surf -- Surf upstream docs: https://docs.asksurf.ai -- Surf publisher: https://asksurf.ai -- BlockRun proxy source: `src/lib/surf.ts` in the BlockRun web repo +- Gateway retirement notice: `curl -s https://blockrun.ai/v1/surf/market/price` (HTTP 410, free) +- Live route catalog: https://blockrun.ai/api/openapi +- Related skills: [`crypto-data`](../crypto-data/SKILL.md) · [`prediction-markets`](../prediction-markets/SKILL.md) · [`rpc`](../rpc/SKILL.md) diff --git a/src/cli/skills.ts b/src/cli/skills.ts index f6ee540..330f88e 100644 --- a/src/cli/skills.ts +++ b/src/cli/skills.ts @@ -127,18 +127,27 @@ export function installSkills(opts: InstallOptions): InstallResult { export interface SkillsArgs { cmd: "list" | "install" | "help"; + /** + * True when the user ASKED for help (`skills`, `skills --help`, `skills + * install -h`); false when `cmd` is "help" only because the subcommand was + * not recognised. Same usage text either way, different exit code. + */ + help: boolean; to?: string; global: boolean; force: boolean; only?: string[]; } -/** Parse everything after the `skills` word. Unknown subcommand → help. */ +/** Parse everything after the `skills` word. Unknown subcommand → help (not requested). */ export function parseSkillsArgs(argv: string[]): SkillsArgs { - const out: SkillsArgs = { cmd: "help", global: false, force: false, only: undefined, to: undefined }; + const out: SkillsArgs = { cmd: "help", help: false, global: false, force: false, only: undefined, to: undefined }; const [first, ...rest] = argv; if (first === "list" || first === "install") out.cmd = first; - else return out; + else { + out.help = first === undefined || first === "--help" || first === "-h"; + return out; + } for (let i = 0; i < rest.length; i++) { const a = rest[i]; @@ -154,7 +163,7 @@ export function parseSkillsArgs(argv: string[]): SkillsArgs { if (!v || v.startsWith("--")) throw new Error("--only requires a comma-separated list of skill names"); out.only = splitList(v); } else if (a.startsWith("--only=")) out.only = splitList(a.slice("--only=".length)); - else if (a === "--help" || a === "-h") out.cmd = "help"; + else if (a === "--help" || a === "-h") { out.cmd = "help"; out.help = true; } else throw new Error(`Unknown option for "skills ${first}": ${a}`); } return out; @@ -221,8 +230,15 @@ export function runSkillsCli(argv: string[], io: { out: (s: string) => void; err } if (args.cmd === "help") { - io.out(skillsUsage()); - return argv.length === 0 || argv[0] === "--help" || argv[0] === "-h" ? 0 : 2; + // Asked for: usage on stdout, exit 0 — `skills install --help && …` in a + // setup script must not abort. Not asked for (unknown subcommand): name + // the problem on stderr, exit 2. + if (args.help) { + io.out(skillsUsage()); + return 0; + } + io.err(`Unknown skills subcommand: ${argv[0]}\n\n${skillsUsage()}`); + return 2; } const skills = listSkills(SKILLS_SOURCE_DIR); diff --git a/src/index.ts b/src/index.ts index e2f90f8..d18646f 100644 --- a/src/index.ts +++ b/src/index.ts @@ -14,7 +14,7 @@ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js" import { initializeMcpServer } from "./mcp-handler.js"; import { warnOnLeakedKeys } from "./utils/key-leak-scanner.js"; import { installBlockrunMcpUserAgent } from "./utils/user-agent.js"; -import { PROFILES } from "./profiles.js"; +import { PROFILES, knownProfileNames, resolveProfileName } from "./profiles.js"; import { runSkillsCli } from "./cli/skills.js"; // Read version from package.json so it can never drift from the published version. @@ -75,8 +75,14 @@ async function checkForUpdate() { }); const data = await resp.json() as { version?: string }; if (data.version && data.version !== VERSION) { + // Everyone who sees this already has the server registered, and + // `claude mcp add` refuses a name that exists — so never print that. + // Registered with @latest (the documented install) a restart is the whole + // upgrade: npx re-resolves the tag on every cold start. Only a pinned + // version needs re-registering, and that is remove-then-add. console.error(`[BlockRun] Update available: v${VERSION} → v${data.version}`); - console.error(`[BlockRun] Run: claude mcp add blockrun -s user -- npx -y @blockrun/mcp@latest`); + console.error(`[BlockRun] Registered with @latest? Restart your MCP client — npx re-resolves on cold start (stale cache: rm -rf ~/.npm/_npx).`); + console.error(`[BlockRun] Pinned to a version? claude mcp remove blockrun -s user, then re-add with @blockrun/mcp@latest. Details: the blockrun-upgrade skill.`); } } catch { // Don't block startup on network issues @@ -98,6 +104,15 @@ async function main() { const transport = new StdioServerTransport(); await server.connect(transport); + // resolveTools falls back to "full" for a name it does not know. Say so: + // a user who typed `--profile tradng` wanted 9 tools and got 20, and the + // "20 tools" startup line alone reads as if the flag was honoured. + const requestedProfile = resolveProfileName(); + if (requestedProfile !== profile) { + console.error( + `[BlockRun] Unknown profile "${requestedProfile}" — exposing the full tool set instead. Known profiles: ${knownProfileNames().join(", ")}`, + ); + } const profileNote = profile === "full" ? `${tools.length} tools` : `profile "${profile}" — ${tools.length} tools: ${tools.join(", ")}`; diff --git a/src/mcp-handler.ts b/src/mcp-handler.ts index 1177f68..60caf61 100644 --- a/src/mcp-handler.ts +++ b/src/mcp-handler.ts @@ -20,7 +20,6 @@ import { registerPriceTool } from "./tools/price.js"; import { registerDexTool } from "./tools/dex.js"; import { registerModalTool } from "./tools/modal.js"; import { registerPhoneTool } from "./tools/phone.js"; -import { registerSurfTool } from "./tools/surf.js"; import { registerRpcTool } from "./tools/rpc.js"; import { registerDefiTool } from "./tools/defi.js"; import { registerPolymarketReadTool, registerPolymarketTool } from "./tools/polymarket.js"; @@ -48,8 +47,22 @@ export function initializeMcpServer( // ledger starts unlimited; the cap is in-memory and resets when the (npx-spawned) // process restarts, so an operator who wants a hard ceiling should set the env. const env = profileArgs?.env ?? process.env; + const rawLimit = env.BLOCKRUN_BUDGET_LIMIT; + const limit = parseBudgetLimitEnv(rawLimit); + // parseBudgetLimitEnv maps anything that is not a finite positive number to + // null — and null here means UNLIMITED. That contract is shared with + // BLOCKRUN_CONFIRM_THRESHOLD and stays; what must not stay is the silence. An + // operator who wrote "5,00", "5 USD", "0" or "-3" believes the hard stop is + // on. Say so once, on stderr (the MCP stdio log channel — stdout is the + // protocol). Unset or blank is the default, not a misconfiguration. + if (rawLimit?.trim() && limit === null) { + console.error( + `[BlockRun] BLOCKRUN_BUDGET_LIMIT="${rawLimit}" is not a positive USD amount — the spend cap is OFF (unlimited). ` + + `Write it as a plain number, e.g. BLOCKRUN_BUDGET_LIMIT=5 or BLOCKRUN_BUDGET_LIMIT=$2.50`, + ); + } const budget: BudgetState = { - limit: parseBudgetLimitEnv(env.BLOCKRUN_BUDGET_LIMIT), + limit, spent: 0, calls: 0, agents: new Map(), @@ -76,7 +89,6 @@ export function initializeMcpServer( dex: () => registerDexTool(server), modal: () => registerModalTool(server, budget), phone: () => registerPhoneTool(server, budget), - surf: () => registerSurfTool(server, budget), rpc: () => registerRpcTool(server, budget), defi: () => registerDefiTool(server, budget), polymarket_read: () => registerPolymarketReadTool(server), diff --git a/src/profiles.ts b/src/profiles.ts index 34ec67c..62bbaa3 100644 --- a/src/profiles.ts +++ b/src/profiles.ts @@ -24,7 +24,6 @@ export type ToolName = | "dex" | "modal" | "phone" - | "surf" | "rpc" | "defi" | "polymarket_read" @@ -35,7 +34,7 @@ export type ToolName = // real ToolName (catches typos). export const ALL_TOOLS = [ "wallet", "chat", "models", "image", "music", "speech", "video", "realface", - "search", "exa", "markets", "price", "dex", "modal", "phone", "surf", "rpc", "defi", + "search", "exa", "markets", "price", "dex", "modal", "phone", "rpc", "defi", "polymarket_read", "polymarket", ] as const satisfies readonly ToolName[]; @@ -57,20 +56,30 @@ export const PROFILES: Record = { // Markets & on-chain data: prediction markets (data + Polymarket trading), // realtime prices, DEX/CEX data, DeFi metrics, and raw RPC, plus the wallet // for balance/funding. - trading: ["wallet", "price", "dex", "markets", "surf", "defi", "rpc", "polymarket_read", "polymarket"], + trading: ["wallet", "price", "dex", "markets", "defi", "rpc", "polymarket_read", "polymarket"], // Web research & analysis: live search, neural search, Surf's news/SQL, // and chat for synthesis, plus wallet and the model catalogue. - research: ["wallet", "models", "chat", "search", "exa", "surf"], + research: ["wallet", "models", "chat", "search", "exa"], // Minimal LLM gateway: just chat + model discovery + wallet. chat: ["wallet", "models", "chat"], }; const DEFAULT_PROFILE = "full"; +// Profile names are case-insensitive and whitespace-tolerant: a JSON client's +// `"args": ["--profile", "trading "]` or `BLOCKRUN_MCP_PROFILE=" Media"` is a +// typo, not a different profile. Blank means "not specified". +function normalizeProfileName(raw: string | undefined): string { + const name = (raw ?? "").trim().toLowerCase(); + return name || DEFAULT_PROFILE; +} + /** * Resolve the active profile name. Precedence: explicit `--profile ` / * `--profile=` CLI flag, then BLOCKRUN_MCP_PROFILE env, then "full". - * An unknown name falls back to "full" (logged by the caller). + * Returns the name as REQUESTED (trimmed, lower-cased); it may not be a known + * profile — `resolveTools` does the fallback and reports both names so the + * caller can log when they differ. */ export function resolveProfileName( argv: string[] = process.argv.slice(2), @@ -78,32 +87,40 @@ export function resolveProfileName( ): string { for (let i = 0; i < argv.length; i++) { const arg = argv[i]; - if (arg === "--profile") return (argv[i + 1] ?? DEFAULT_PROFILE).toLowerCase(); - if (arg.startsWith("--profile=")) return arg.slice("--profile=".length).toLowerCase(); + if (arg === "--profile") return normalizeProfileName(argv[i + 1]); + if (arg.startsWith("--profile=")) return normalizeProfileName(arg.slice("--profile=".length)); } - if (env.BLOCKRUN_MCP_PROFILE) return env.BLOCKRUN_MCP_PROFILE.toLowerCase(); - return DEFAULT_PROFILE; + return normalizeProfileName(env.BLOCKRUN_MCP_PROFILE); +} + +/** The profile names `--profile` accepts, for help text and the unknown-name warning. */ +export function knownProfileNames(): string[] { + return Object.keys(PROFILES); } /** * Resolve a profile name to the concrete set of tools to register. * Returns the canonical profile name actually used (after unknown-name - * fallback) alongside the tool list, so the server can log it accurately. + * fallback) alongside the tool list, plus the name that was requested, so the + * server can log accurately — and can say so when the two differ, because a + * misspelt `--profile tradng` otherwise loads all 20 schemas in silence for a + * user who asked for 9. */ export function resolveTools( argv?: string[], env?: NodeJS.ProcessEnv, -): { profile: string; tools: Set } { +): { profile: string; tools: Set; requested: string } { const requested = resolveProfileName(argv, env); // Use hasOwn so inherited Object.prototype members ("constructor", // "__proto__", …) are treated as unknown names and fall back to "full" // instead of resolving to a non-iterable function and crashing at startup. const spec = Object.hasOwn(PROFILES, requested) ? PROFILES[requested] : undefined; if (!spec) { - return { profile: DEFAULT_PROFILE, tools: new Set(ALL_TOOLS) }; + return { profile: DEFAULT_PROFILE, tools: new Set(ALL_TOOLS), requested }; } return { profile: requested, tools: new Set(spec === "all" ? ALL_TOOLS : spec), + requested, }; } diff --git a/src/tools/chat-anthropic.ts b/src/tools/chat-anthropic.ts index 1161787..de242dc 100644 --- a/src/tools/chat-anthropic.ts +++ b/src/tools/chat-anthropic.ts @@ -67,21 +67,47 @@ const OUTPUT_QUOTE_FACTOR = 0.1; const MESSAGE_TOKEN_OVERHEAD = 20; const MIN_BASE_USD = 0.001; +/** + * Map the id the gateway ECHOES onto the id the catalogue KEYS on. + * + * /v1/messages echoes the upstream Anthropic id (blockrun's ANTHROPIC_MODEL_MAP), + * not the id that was requested: "anthropic/claude-fable-5.1" comes back as + * "claude-fable-5-1", "anthropic/claude-haiku-4.5" as + * "claude-haiku-4-5-20251001". Three differences, undone in order: the vendor + * prefix is missing, a -YYYYMMDD snapshot date may be appended, and the minor + * version is dashed where the catalogue spells it dotted. Nothing else is + * touched, so an id this does not recognise misses the table and books null. + */ +function catalogueKeyForEcho(model: string): string { + let id = model.trim(); + if (!id.startsWith("anthropic/")) id = `anthropic/${id}`; + id = id.replace(/-\d{8}$/, ""); + id = id.replace(/^(anthropic\/claude-[a-z]+-\d+)-(\d+)$/, "$1.$2"); + return id; +} + export function anthropicCallCost( model: string, promptChars: number, maxTokens: number, ): number | null { - // The catalog keys on the prefixed id; the response echoes a bare one - // ("claude-opus-5"), sometimes with a date suffix. - const id = model.startsWith("anthropic/") ? model : `anthropic/${model}`; + const id = catalogueKeyForEcho(model); // hasOwn, not `??` — see the note in estimateChatCost: an inherited // Object.prototype member would pass the null check and poison the arithmetic. // (`id` is always prefixed with "anthropic/" here, so it cannot BE a prototype // key; guarded anyway so the pattern is uniform wherever these tables are read.) - const rate = Object.hasOwn(CHAT_PRICE_PER_MTOKEN, id) - ? CHAT_PRICE_PER_MTOKEN[id] - : Object.entries(CHAT_PRICE_PER_MTOKEN).find(([k]) => id.startsWith(k))?.[1]; + // + // Exact match ONLY. This used to fall back to a startsWith prefix match, meant + // for date-suffixed echoes — but the gateway echoes DASHED upstream ids, so + // the prefix never matched a dated echo at all (claude-haiku-4-5-20251001 + // does not start with anthropic/claude-haiku-4.5) and every one of them + // silently booked the pre-call estimate. Its one live use was matching a + // VERSION suffix: claude-fable-5-1 booked claude-fable-5's row. Right by + // coincidence (both $10/$50) — and a sibling priced differently from its + // major would have booked the wrong number with no signal, because the + // "null -> estimate" fallback cannot engage once a rate WAS found. On this + // path the table is the ledger, so a borrowed rate is a wrong budget.spent. + const rate = Object.hasOwn(CHAT_PRICE_PER_MTOKEN, id) ? CHAT_PRICE_PER_MTOKEN[id] : undefined; if (!rate) return null; const inputTokens = Math.ceil(promptChars / GATEWAY_CHARS_PER_TOKEN_OBSERVED) + MESSAGE_TOKEN_OVERHEAD; diff --git a/src/tools/chat.ts b/src/tools/chat.ts index 11f4f6a..c78e0a2 100644 --- a/src/tools/chat.ts +++ b/src/tools/chat.ts @@ -14,6 +14,7 @@ import { FREE_TIER_MAX_PROMPT_CHARS, CHAT_PRICE_PER_MTOKEN, DEFAULT_CHAT_PRICE, + FREE_CHAT_MODELS, canonicalChatModel, TIER_WORST_PRICE, GATEWAY_CHARS_PER_TOKEN, @@ -61,6 +62,13 @@ export function freeTierTruncationNote(promptChars: number, model: string): stri // `nvidia/gpt-oss-120b`, and the bare spelling truncates identically — a // startsWith check on the raw string let the silent-truncation warning go // silent, which is the one failure this function exists to make loud. + // + // Still a VENDOR test, on purpose, unlike the $0 classifier (FREE_CHAT_MODELS): + // the 128 KiB cap was measured on the NVIDIA free path and nowhere else. The + // cohere/poolside free models are unmeasured, and a warning that says "a + // third of your prompt was dropped" must not be extended to a path where it + // may not have been — that would push agents off a working $0 path onto paid + // USDC on a false premise, the exact harm the byte-vs-char fix removed. if (!canonicalChatModel(model).startsWith("nvidia/")) return null; // paid models scale past this if (promptChars <= FREE_TIER_MAX_PROMPT_CHARS) return null; const keptPct = Math.round((FREE_TIER_MAX_PROMPT_CHARS / promptChars) * 100); @@ -108,7 +116,11 @@ export function estimateChatCost( // nvidia check included — has to run on the catalog spelling. const canonical = model ? canonicalChatModel(model) : undefined; if (canonical) { - if (canonical.startsWith("nvidia/")) return 0; // genuinely free, whatever the mode + // Membership, not vendor: the catalogue bills cohere/north-mini-code and + // poolside/laguna-xs-2.1 at $0 too, and a `startsWith("nvidia/")` here + // reserved the $5/$30 default for them — an exhausted budget refused a + // free call. See FREE_CHAT_MODELS for the sweep that keeps the set honest. + if (FREE_CHAT_MODELS.has(canonical)) return 0; // genuinely free, whatever the mode } else if (mode === "free") { return 0; // no model to override it — resolves to the free tier } @@ -117,7 +129,19 @@ export function estimateChatCost( // not max_tokens — is the dominant cost driver on the native claude-* path. // Fold it into the reserved output size so the gate can't be bypassed by a // tiny max_tokens + a huge budget_tokens. - const out = Math.max((maxTokens ?? 1024) + (thinkingBudget ?? 0), 256); + // + // ONLY there, though. `thinking` is forwarded on the native claude-* path and + // nowhere else — see the isAnthropicModel dispatch in the handler; the + // OpenAI-compat paths build their options from max_tokens/temperature/ + // response_format/stop, exactly as the schema's "Ignored for non-Claude + // models" promises. Folding it unconditionally reserved ~$18 for + // mode:"powerful" + a 100k budget (gpt-5.4-pro output at $180/M) on a call + // that settles at cents: a spurious refusal for a delegated agent, and a + // wrong "Estimated: $X" put in front of a human under BLOCKRUN_CONFIRM_SPEND. + // Same classifier as the dispatch, run on the canonical id, so the two agree + // for the prefixed and the bare claude-* spelling alike. + const thinkingOut = canonical && isAnthropicModel(canonical) ? (thinkingBudget ?? 0) : 0; + const out = Math.max((maxTokens ?? 1024) + thinkingOut, 256); // Reserve at the REAL rate of what this call can settle at — the named model's // own price, or the most expensive member of the tier it will route through. @@ -215,6 +239,32 @@ async function withSettledCost( } } +/** + * The error text for a call that SETTLED and then failed. + * + * x402 settles on the 200, before the body is read, and every paid path streams, + * so a stall or an in-band error event arrives with the money already gone. + * withSettledCost books it (onSettledThrow); this is the sentence that tells the + * CALLER. Without it the text was "Error: stream stalled: no data from the + * gateway for 120s" — indistinguishable from a free failure, so the obvious next + * step (retry) settled a second payment. The routing loop has said this since + * 0.40.1; the explicit-model and multi-turn paths, which by construction fail + * only after settlement, never did. + * + * formatError runs on the BARE error and the note is appended afterwards, on + * purpose: formatError classifies on keywords, and this note contains the word + * "payment", which its funding branch reads as an empty wallet. Fed the combined + * text, the routing loop's version ended in "your wallet needs funding" — the + * exact wrong advice for a call that just paid. + */ +function settledThenFailedText(error: unknown, settledUsd: number, tail: string): string { + return ( + `${formatError(extractErrorMessage(error))}\n\nNote: payment had already settled when this failed, ` + + `so the charge stands ($${settledUsd.toFixed(6)}) and it has been recorded against your budget. ${tail}` + ); +} +const RETRY_CHARGES_AGAIN = 'Retrying will incur a second charge — check blockrun_wallet action:"report" first.'; + export function registerChatTool(server: McpServer, budget: BudgetState): void { server.registerTool( "blockrun_chat", @@ -227,7 +277,7 @@ Notable modes: - mode:"coding" → Claude Opus 5, GPT-5.3-codex, Kimi K3, Grok Build, GLM-5.2 - mode:"cheap" → deepseek-v4-pro, Qwen3.7 Flash, MiniMax M3, Tencent Hy3 - mode:"glm" → Zhipu GLM-5 / 5.2 / 5.1 / 5-Turbo (strong at coding) -- mode:"free" → NVIDIA models (no cost) +- mode:"free" → free models (no cost) Pick directly: model:"anthropic/claude-opus-5", model:"moonshot/kimi-k3", model:"openai/gpt-5.6-sol", model:"xai/grok-4.5", model:"nvidia/gpt-oss-120b" (free). @@ -236,7 +286,7 @@ Run blockrun_models to see all available models with pricing.`, inputSchema: { message: z.string().describe("Your message to the AI"), model: z.string().optional().describe("Specific model ID (e.g., 'moonshot/kimi-k3', 'openai/gpt-5.6-sol', 'zai/glm-5')"), - mode: z.enum(["fast", "balanced", "powerful", "cheap", "reasoning", "free", "coding", "glm"]).optional().describe("Routing mode: powerful/reasoning = frontier models (Opus 5, GPT-5.6-sol, Kimi K3), coding = code-specialized, glm = Zhipu GLM (great for coding), cheap = budget models, free = NVIDIA only (ignored if model specified)"), + mode: z.enum(["fast", "balanced", "powerful", "cheap", "reasoning", "free", "coding", "glm"]).optional().describe("Routing mode: powerful/reasoning = frontier models (Opus 5, GPT-5.6-sol, Kimi K3), coding = code-specialized, glm = Zhipu GLM (great for coding), cheap = budget models, free = $0 models (ignored if model specified)"), system: z.string().optional().describe("Optional system prompt"), max_tokens: z.number().optional().default(1024).describe("Max tokens in response"), temperature: z.number().optional().default(1).describe("Creativity 0-2"), @@ -346,6 +396,8 @@ Run blockrun_models to see all available models with pricing.`, ...messages, { role: "user" as const, content: message }, ]; + // USDC that left the wallet before the failure, if any (see settledThenFailedText). + let settledOnFailure = 0; try { // The SDK types ChatMessage.content as string-only, but the gateway // forwards `messages` verbatim and accepts image_url content arrays @@ -375,7 +427,10 @@ Run blockrun_models to see all available models with pricing.`, stop, }); return r.choices?.[0]?.message?.content || ""; - }, (usd) => recordActualSpend(budget, usd, estimatedCost, agent_id)); + }, (usd) => { + recordActualSpend(budget, usd, estimatedCost, agent_id); + settledOnFailure = usd; + }); recordActualSpend(budget, settledUsd, estimatedCost, agent_id); const note = freeTierTruncationNote(promptChars, targetModel); return { @@ -383,13 +438,22 @@ Run blockrun_models to see all available models with pricing.`, structuredContent: { model_used: targetModel, response: reply, message_count: fullMessages.length, ...(note ? { truncated: true } : {}) }, }; } catch (error) { - return { content: [{ type: "text", text: formatError(extractErrorMessage(error)) }], isError: true }; + return { + content: [{ + type: "text", + text: settledOnFailure > 0 + ? settledThenFailedText(error, settledOnFailure, RETRY_CHARGES_AGAIN) + : formatError(extractErrorMessage(error)), + }], + isError: true, + }; } } // If specific model provided, use it directly — streamed when the client // supports it (same 524 rationale as the multi-turn path above). if (model) { + let settledOnFailure = 0; try { const { result: response, settledUsd } = await withSettledCost(llm(), async () => { const client = llm(); @@ -406,12 +470,20 @@ Run blockrun_models to see all available models with pricing.`, responseFormat, stop, }); - }, (usd) => recordActualSpend(budget, usd, estimatedCost, agent_id)); + }, (usd) => { + recordActualSpend(budget, usd, estimatedCost, agent_id); + settledOnFailure = usd; + }); recordActualSpend(budget, settledUsd, estimatedCost, agent_id); return { content: [{ type: "text", text: `${response}${freeTierTruncationNote(promptChars, model) ?? ""}` }] }; } catch (error) { return { - content: [{ type: "text", text: formatError(extractErrorMessage(error)) }], + content: [{ + type: "text", + text: settledOnFailure > 0 + ? settledThenFailedText(error, settledOnFailure, RETRY_CHARGES_AGAIN) + : formatError(extractErrorMessage(error)), + }], isError: true, }; } @@ -485,19 +557,27 @@ Run blockrun_models to see all available models with pricing.`, } } + // Say it plainly: the payment settled before the failure, so the charge + // stands and no fallback was attempted. An agent that reads "failed" as + // "free" would retry in a loop and pay each time. (Free models settle $0, + // so the deadline case below can never also be a settled one.) + if (settledOnFailure > 0) { + return { + content: [{ + type: "text", + text: settledThenFailedText(lastError, settledOnFailure, "No fallback model was tried — retrying will incur a second charge."), + }], + isError: true, + }; + } // Distinguish "every model rejected" from "we ran out of time" — they need // different things from the caller (retry vs. pick a paid model), and a bare // last-error would have blamed whichever model happened to be slowest. const errorMessage = deadlineHit - ? `The free tier did not answer within ${Math.round(FREE_TIER_DEADLINE_MS / 1000)}s. Free NVIDIA capacity is usually saturated when this happens — retry shortly, or pass an explicit model (or a paid mode) to skip the free tier.` - : settledOnFailure > 0 - // Say it plainly: the payment settled before the failure, so the - // charge stands and no fallback was attempted. An agent that reads - // "failed" as "free" would retry in a loop and pay each time. - ? `${extractErrorMessage(lastError)}\n\nNote: payment had already settled when this failed, so the charge stands ($${settledOnFailure.toFixed(6)}) and it has been recorded against your budget. No fallback model was tried — retrying will incur a second charge.` - : lastError - ? extractErrorMessage(lastError) - : "All models failed"; + ? `The free tier did not answer within ${Math.round(FREE_TIER_DEADLINE_MS / 1000)}s. Free-tier capacity is usually saturated when this happens — retry shortly, or pass an explicit model (or a paid mode) to skip the free tier.` + : lastError + ? extractErrorMessage(lastError) + : "All models failed"; return { content: [{ type: "text", text: formatError(errorMessage) }], isError: true, diff --git a/src/tools/defi.ts b/src/tools/defi.ts index 2146cfb..dca4415 100644 --- a/src/tools/defi.ts +++ b/src/tools/defi.ts @@ -32,19 +32,19 @@ export function registerDefiTool(server: McpServer, budget: BudgetState): void { { description: `DeFi fundamentals via DefiLlama — protocol TVL, chain TVL, yield pools (APY), token prices. Pays per call in USDC, no API key. -Paths (GET only): -- protocols ($0.007 charged) — all DeFi protocols ranked by TVL -- protocol/{slug} ($0.007 charged) — one protocol's TVL history + chain breakdown, e.g. protocol/aave-v3 -- chains ($0.007 charged) — TVL by chain -- yields ($0.007 charged) — yield pools with APY + TVL (large; filter client-side) -- prices/{coins} ($0.003 charged) — token prices, coins like 'base:0x833589...,coingecko:ethereum' +Paths (GET only; price = base + the gateway's flat tx fee, $0.001 today — we reserve $0.002; the 402 header carries the exact charge): +- protocols ($0.005 base) — all DeFi protocols ranked by TVL +- protocol/{slug} ($0.005 base) — one protocol's TVL history + chain breakdown, e.g. protocol/aave-v3 +- chains ($0.005 base) — TVL by chain +- yields ($0.005 base) — yield pools with APY + TVL (large; filter client-side) +- prices/{coins} ($0.001 base) — token prices, coins like 'base:0x833589...,coingecko:ethereum' Examples: blockrun_defi({ path: "protocol/uniswap-v3" }) blockrun_defi({ path: "prices/coingecko:bitcoin,coingecko:ethereum" }) blockrun_defi({ path: "chains" }) -Use blockrun_price (free) for plain spot quotes, blockrun_dex (free) for DEX pairs, blockrun_surf for labeled on-chain data — this tool is for protocol/TVL/yield fundamentals.`, +Use blockrun_price (free) for plain spot quotes, blockrun_dex (free) for DEX pairs — this tool is for protocol/TVL/yield fundamentals.`, annotations: TOOL_ANNOTATIONS.readOnlyOpenWorld, inputSchema: { path: z.string().describe("Endpoint under /v1/defillama/, e.g. 'protocols', 'protocol/aave-v3', 'chains', 'yields', 'prices/coingecko:ethereum'"), diff --git a/src/tools/dex.ts b/src/tools/dex.ts index ed77431..b7dbd8a 100644 --- a/src/tools/dex.ts +++ b/src/tools/dex.ts @@ -4,6 +4,30 @@ import { TOOL_ANNOTATIONS } from "../tool-annotations.js"; import { z } from "zod"; import { fetchWithTimeout } from "../utils/http.js"; +/** + * What DexScreener's `/latest/dex/tokens/{addresses}` accepts: one or more + * (up to 30, comma-separated) token addresses — 0x…40-hex on EVM chains, + * base58 on Solana, and the longer forms other chains use (Sui coin types like + * `0x2::sui::SUI`, TON, Aptos). Rather than enumerate those per chain, allow + * the characters addresses are made of and nothing that URL syntax gives a + * meaning to: no `/ ? # % & +`, no whitespace, and no segment that starts with + * a dot (so `.` and `..` cannot travel up the path). `:` is a legal path + * character (RFC 3986 pchar), so no percent-encoding is needed once the shape + * is enforced. + */ +const TOKEN_ADDRESS_RE = /^[A-Za-z0-9][A-Za-z0-9_:.-]{1,199}$/; +const MAX_TOKEN_ADDRESSES = 30; + +/** + * Split and validate the `token` argument. Returns the cleaned addresses, or + * null when any item fails the shape check. Exported for tests. + */ +export function parseTokenAddresses(raw: string): string[] | null { + const items = raw.split(",").map((s) => s.trim()).filter((s) => s.length > 0); + if (items.length === 0 || items.length > MAX_TOKEN_ADDRESSES) return null; + return items.every((s) => TOKEN_ADDRESS_RE.test(s)) ? items : null; +} + export function registerDexTool(server: McpServer): void { server.registerTool( "blockrun_dex", @@ -33,7 +57,24 @@ Examples: let searchTerm = query || symbol || ""; if (token) { - url = `https://api.dexscreener.com/latest/dex/tokens/${token}`; + // The caller's string goes into the URL PATH. Validate the shape + // instead of splicing it raw: `../search?q=pepe` or `abc#x` would + // otherwise rewrite the request and return another endpoint's + // answer (or nothing) labelled as token data. + const addresses = parseTokenAddresses(token); + if (!addresses) { + return { + content: [{ + type: "text", + text: + `Invalid token address: ${JSON.stringify(token)}. Expected a contract or mint address ` + + `(0x… on EVM chains, base58 on Solana), or up to ${MAX_TOKEN_ADDRESSES} of them separated by commas. ` + + `To search by name or symbol use query instead.`, + }], + isError: true, + }; + } + url = `https://api.dexscreener.com/latest/dex/tokens/${addresses.join(",")}`; } else if (searchTerm) { url = `https://api.dexscreener.com/latest/dex/search?q=${encodeURIComponent(searchTerm)}`; } else { diff --git a/src/tools/exa.ts b/src/tools/exa.ts index f342d90..43b422d 100644 --- a/src/tools/exa.ts +++ b/src/tools/exa.ts @@ -44,10 +44,11 @@ export function registerExaTool(server: McpServer, budget: BudgetState): void { description: `Neural web search via Exa — understands meaning, not just keywords. Great for research. Common paths (all POST, body shapes documented in the exa-research skill): -- search — body: { query, numResults?, category?, includeDomains?, excludeDomains? } ($0.012/call charged) -- answer — body: { query } ($0.012/call charged) -- contents — body: { urls: [...] } ($0.002/URL + $0.002 fee, up to 100) -- find-similar — body: { url, numResults? } ($0.012/call charged) +- search — body: { query, numResults?, category?, includeDomains?, excludeDomains? } ($0.010 base + tx fee) +- answer — body: { query } ($0.010 base + tx fee) +- contents — body: { urls: [...] } ($0.002/URL + ONE tx fee, up to 100) +- find-similar — body: { url, numResults? } ($0.010 base + tx fee) +Tx fee = the gateway's flat network fee, $0.001 today (we reserve $0.002); the 402 header carries the exact charge. Categories for search: "news", "research paper", "company", "tweet", "github", "pdf". diff --git a/src/tools/image.ts b/src/tools/image.ts index da95ab7..8fe58b9 100644 --- a/src/tools/image.ts +++ b/src/tools/image.ts @@ -3,7 +3,7 @@ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; import { TOOL_ANNOTATIONS } from "../tool-annotations.js"; import { z } from "zod"; import { PaymentError } from "@blockrun/llm"; -import { reserveBudget, recordSpending, recordActualSpend, reReserveIfHigher, BudgetExceededError } from "../utils/budget.js"; +import { BudgetExceededError, assertQuoteNearEstimate, reReserveIfHigher, recordActualSpend, recordSpending, reserveBudget } from "../utils/budget.js"; import { withTxFee } from "../utils/tx-fee.js"; import { formatError } from "../utils/errors.js"; import { launchTopUp } from "../utils/onramp.js"; @@ -15,7 +15,7 @@ import { solanaPaidPost } from "../utils/solana-402.js"; import { isBlockedFetchHostResolved } from "../utils/ssrf.js"; import { shouldInline, buildInlineImageBlock } from "../utils/inline-image.js"; import { confirmSpend } from "../utils/confirm-spend.js"; -import { readFile, writeFile } from "node:fs/promises"; +import { readFile, realpath, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { randomBytes } from "node:crypto"; @@ -35,8 +35,28 @@ const IMAGE_EXT_MIME: Record = { webp: "image/webp", }; +/** A source image or mask, normalized for the gateway. */ +export interface ResolvedImageRef { + /** What goes in the request body. */ + dataUri: string; + /** + * The REAL filesystem path this data URI was read from — after + * fs.realpath, so a symlink is reported as its target — or undefined for a + * data: URI or an http(s) URL. Callers surface it wherever a human looks + * before the call leaves the machine (the confirmSpend label): the model + * names the file, and "edit ~/Pictures/IMG_1234.jpg" must not look exactly + * like any other $0.05 edit. + */ + localPath?: string; +} + +/** Data-URI form only; see resolveImageRef for the local path as well. */ export async function toImageDataUri(ref: string): Promise { - if (ref.startsWith("data:image/")) return ref; + return (await resolveImageRef(ref)).dataUri; +} + +export async function resolveImageRef(ref: string): Promise { + if (ref.startsWith("data:image/")) return { dataUri: ref }; if (/^https?:\/\//i.test(ref)) { const ctrl = new AbortController(); @@ -78,21 +98,26 @@ export async function toImageDataUri(ref: string): Promise { if (buffer.byteLength > REFERENCE_IMAGE_MAX_BYTES) { throw new Error(`image too large: ${(buffer.byteLength / 1e6).toFixed(1)}MB > ${REFERENCE_IMAGE_MAX_BYTES / 1e6}MB cap`); } - return `data:${mime};base64,${buffer.toString("base64")}`; + return { dataUri: `data:${mime};base64,${buffer.toString("base64")}` }; } finally { clearTimeout(timeout); } } - // Treat as a local file path. + // Treat as a local file path. Deliberately NOT restricted to cwd or tmpdir — + // "edit ~/Downloads/photo.png" from a Desktop session whose cwd is `/` is the + // documented use. What we do owe the user is the truth about which file is + // about to leave: resolve through realpath so the label carries the target + // of a symlink, not whatever innocent name it was given. const ext = ref.split(".").pop()?.toLowerCase() ?? ""; const mime = IMAGE_EXT_MIME[ext]; if (!mime) throw new Error(`unsupported image extension ".${ext}"; use png/jpg/jpeg/gif/webp`); - const buffer = await readFile(ref); + const localPath = await realpath(ref); + const buffer = await readFile(localPath); if (buffer.byteLength > REFERENCE_IMAGE_MAX_BYTES) { throw new Error(`image too large: ${(buffer.byteLength / 1e6).toFixed(1)}MB > ${REFERENCE_IMAGE_MAX_BYTES / 1e6}MB cap; resize or crop first`); } - return `data:${mime};base64,${buffer.toString("base64")}`; + return { dataUri: `data:${mime};base64,${buffer.toString("base64")}`, localPath }; } // Base (1024x1024) prices, mirroring the live /v1/images/models catalog. @@ -321,6 +346,9 @@ Source images and masks accept a base64 data URI, an http(s) URL, or a local fil // consumed at the shared charge site after the spend confirmation. let normalizedImage: string | string[] | undefined; let normalizedMask: string | undefined; + // Real paths of every local file read for this edit (sources, then the + // mask), for the confirm label — see resolveImageRef. + const localFiles: string[] = []; // Validate the edit action up front (before estimating/charging). if (action === "edit") { @@ -359,9 +387,15 @@ Source images and masks accept a base64 data URI, an http(s) URL, or a local fil } } try { - const dataUris = await Promise.all(sourceImages.map(toImageDataUri)); + const resolved = await Promise.all(sourceImages.map(resolveImageRef)); + const dataUris = resolved.map((r) => r.dataUri); normalizedImage = dataUris.length === 1 ? dataUris[0] : dataUris; - if (mask) normalizedMask = await toImageDataUri(mask); + for (const r of resolved) if (r.localPath) localFiles.push(r.localPath); + if (mask) { + const m = await resolveImageRef(mask); + normalizedMask = m.dataUri; + if (m.localPath) localFiles.push(m.localPath); + } } catch (e) { return { content: [{ type: "text", text: formatError(`Could not load source image: ${e instanceof Error ? e.message : String(e)}`) }], @@ -385,9 +419,13 @@ Source images and masks accept a base64 data URI, an http(s) URL, or a local fil // Confirm the spend before charging (elicitation; user can approve // once, approve all for the session, or decline to abort). No-ops on // clients without elicitation or when disabled via env. + // The label names every local file that is about to leave the + // machine in the request body — the dialog is the one moment a human + // sees the call before the bytes go, and a prompt-injected + // "edit ~/Pictures/IMG_1234.jpg" must not read like any other edit. const confirm = await confirmSpend(server, { usd: estimatedCost, - label: `${action === "edit" ? "image edit" : "image"} · ${selectedModel}`, + label: `${action === "edit" ? "image edit" : "image"} · ${selectedModel}${localFiles.length ? ` · reads ${localFiles.join(", ")}` : ""}`, }); if (!confirm.ok) { return { content: [{ type: "text", text: confirm.reason || "Charge cancelled." }] }; @@ -457,7 +495,14 @@ Source images and masks accept a base64 data URI, an http(s) URL, or a local fil // table, so the real quote can exceed what we reserved. Re-reserve // the true amount against the cap BEFORE the transfer is signed // (mirrors blockrun_video); throwing here aborts before any payment. - onQuote: (quotedUsd) => { + onQuote: (quotedUsd, quoteDetails) => { + // WHAT was quoted before how much — a substituted or repriced + // model is refused unsigned (see assertQuoteNearEstimate). + assertQuoteNearEstimate(quotedUsd, estimatedCost, { + what: `${selectedModel} image`, + quotedFor: quoteDetails?.resource?.description, + hint: `Retry on Base (blockrun_wallet action:"chain" chain:"base") or pick another model.`, + }); gate = reReserveIfHigher(budget, gate, agent_id, estimatedCost, quotedUsd); if (!gate.allowed) { throw new BudgetExceededError(`${gate.reason}. Use blockrun_wallet action:"report" to see usage or action:"delegate" to increase agent budget.`); diff --git a/src/tools/markets.ts b/src/tools/markets.ts index 3695126..5f6ceea 100644 --- a/src/tools/markets.ts +++ b/src/tools/markets.ts @@ -10,7 +10,7 @@ import { extractErrorMessage, formatError } from "../utils/errors.js"; import { hasPathTraversal } from "../utils/path-safety.js"; import type { BudgetState } from "../types.js"; import { TOOL_ANNOTATIONS } from "../tool-annotations.js"; -import { validateMarketRequest } from "../utils/markets-validation.js"; +import { describeDegradedSportsFailure, isDegradedSportsPath, validateMarketRequest } from "../utils/markets-validation.js"; // What x402 CHARGES, which is not the 402's JSON `price` field. That field is the // BASE ($0.0075); the charge is base + a $0.002 flat transaction fee, and it lives @@ -33,12 +33,7 @@ export function registerMarketsTool(server: McpServer, budget: BudgetState): voi server.registerTool( "blockrun_markets", { - description: `Prediction market + derivatives data via Predexon aggregator. Flat $0.0095/call (every endpoint) — $0.0075 base + $0.002 tx fee. - -CANONICAL CROSS-VENUE (Tier 1) — Predexon v2 unified data layer: -- markets — list canonical market/question containers with cross-venue Predexon IDs -- outcomes/:predexon_id — resolve a canonical outcome ID to its market context + venue listings - Filter with ?venue=polymarket|kalshi|limitless|opinion|predictfun, ?status=, ?category=, ?league=, ?event_id=, ?pagination_key= + description: `Prediction market + derivatives data via Predexon aggregator. Flat $0.0075 base per call (every endpoint) plus the gateway's network fee — $0.001 on Base today, none quoted on Solana; the 402 header carries the exact charge (we reserve $0.0095). POLYMARKET (Tier 1): - polymarket/events, polymarket/markets — list events/markets (filter, sort, paginate) @@ -67,19 +62,14 @@ WALLET IDENTITY & CLUSTERING (Tier 2) — cross-context labels + on-chain relati - polymarket/wallet/identities — POST { addresses: [...] } for bulk lookup (up to 200 wallets) - polymarket/wallet/:address/cluster — discover wallets connected via on-chain transfers + identity proofs -SPORTS (Tier 1): -- sports/categories — list available sports categories -- sports/markets — list sports markets grouped by game (filter ?league=, ?sport_type=, ?status=, ?venue=) -- sports/markets/:game_id — single sports game with all venue outcomes -- sports/outcomes/:predexon_id — equivalent sports outcomes across venues for a Predexon ID +SPORTS — sports/* (categories, markets, markets/:game_id, outcomes/:predexon_id) DEGRADED, do not call: Predexon 500 on every call since 2026-08-04; the gateway releases the payment on that upstream 500. Use markets/search { q: "NBA" } or polymarket/events { search: "NBA" } instead — no live route takes a "league" param. KALSHI: kalshi/markets, kalshi/trades, kalshi/orderbooks LIMITLESS / OPINION / PREDICT.FUN: {platform}/markets, {platform}/orderbooks BINANCE FUTURES: binance/candles/:symbol, binance/ticks/:symbol CROSS-PLATFORM: -- matching-markets, matching-markets/pairs — equivalent markets across Polymarket+Kalshi -- markets/search — search across all platforms in one call +- markets/search — search every venue in one call (search term is "q"). The only canonical-layer route left: markets, markets/listings, outcomes/:id and matching-markets(/pairs) were removed upstream 2026-08-04 and 404 before payment. REQUEST CONTRACTS: - Discover current markets with markets/search (its search term is "q"), then resolve the chosen Polymarket market with polymarket/markets/keyset and condition_id. @@ -121,7 +111,13 @@ Pass query params via 'params' (GET). Use 'body' only for POST endpoints (e.g. p // Human-in-the-loop (BLOCKRUN_CONFIRM_SPEND=on): ask before signing. A // decline returns here — nothing is sent, and the finally releases the // reservation. No-ops when off, sub-threshold, or unsupported by the client. - const confirm = await confirmSpend(server, { usd: estimatedCost, label: `markets · ${path}` }); + // sports/* is reserved and confirmed like any other route: if Predexon + // recovers, the call WILL settle $0.0095, and an un-reserved settle is the + // worse failure. The label says why the prompt will probably be moot. + const confirm = await confirmSpend(server, { + usd: estimatedCost, + label: isDegradedSportsPath(path) ? `markets · ${path} (degraded upstream — likely fails, usually uncharged)` : `markets · ${path}`, + }); if (!confirm.ok) return { content: [{ type: "text", text: confirm.reason ?? "Charge cancelled." }] }; // rawGet/rawPost rather than the SDK's pm()/pmQuery(): those are one-line // wrappers over exactly `/v1/pm/${path}` on the same raw methods @@ -143,8 +139,13 @@ Pass query params via 'params' (GET). Use 'body' only for POST endpoints (e.g. p gate.release(); } } catch (err) { + const message = extractErrorMessage(err); + // A sports/* 5xx is the known Predexon outage, not a blip, and the + // gateway released the payment — say so instead of "after payment … + // try again in a few minutes" (blockrun-mcp#132). + const degraded = describeDegradedSportsFailure(path, message); return { - content: [{ type: "text", text: formatError(extractErrorMessage(err)) }], + content: [{ type: "text", text: degraded ?? formatError(message) }], isError: true, }; } diff --git a/src/tools/music.ts b/src/tools/music.ts index efdb0cd..648d469 100644 --- a/src/tools/music.ts +++ b/src/tools/music.ts @@ -12,7 +12,7 @@ import { pollDeadline, pollTimeoutFor } from "../utils/poll.js"; import type { BudgetState } from "../types.js"; import { getApiBase, getChain, getOrCreateWalletKey, resolveGatewayUrl } from "../utils/wallet.js"; import { isApiKeyMode } from "../utils/auth.js"; -import { apiKeyAsyncPost } from "../utils/api-key-call.js"; +import { apiKeyAsyncPost, BilledJobError } from "../utils/api-key-call.js"; import { privateKeyToAccount } from "viem/accounts"; import { createPaymentPayload, @@ -93,8 +93,9 @@ export function registerMusicTool(server: McpServer, budget: BudgetState): void Generates a full-length ~3 minute MP3 track. Takes 1-3 minutes to complete. The tool submits the job and, for slower tracks, polls until it is ready; payment -settles only when a finished track is returned — if it fails or times out, you -are not charged. +settles only when a finished track is returned — if it fails you are not +charged; if this client gives up while a paid request is still in flight the +gateway may still settle, and the error text says so. Model: minimax/music-2.5+ ($0.1575/track, up to ~4 min) @@ -117,6 +118,16 @@ Returns a permanent BlockRun-hosted URL.`, // Reserve the estimate up front so concurrent calls can't each pass a // stale budget; release in finally once the call settles or fails. let gate: ReturnType | undefined; + // Visible to the catch, which has to book money that moved without a + // result: the account rail bills at submit, and a Base request aborted in + // flight can still settle server-side. Every give-up also names the job. + let jobId: string | undefined; + let quotedUsd: number | null = null; + // True while a Base request carrying the payment header — the submit, + // which can settle inline, or a poll — has been issued and has not + // answered. A poll that rejects leaves it true: that request may still be + // settling on the gateway, which does not stop on disconnect. + let paidRequestInFlight = false; try { // NO CHAIN GUARD. This tool refused every Solana call until 2026-09-05 // ("settles on Base only"), which stopped being true well before that: @@ -198,6 +209,7 @@ Returns a permanent BlockRun-hosted URL.`, const paymentRequired = parsePaymentRequired(prHeader); const details = extractPaymentDetails(paymentRequired); + quotedUsd = amountToUsd(details.amount); // validBefore is counted from HERE, so the authorization deadline has to // be stamped here too — not after submit, which can burn up to 95s. @@ -224,6 +236,7 @@ Returns a permanent BlockRun-hosted URL.`, // Step 2: submit with payment. Fast tracks complete inline (200); slower // ones (MiniMax music is 1-3 min) return 202 + poll_url — the server // verified the payment but does NOT settle until a completed poll. + paidRequestInFlight = true; const submitResp = await fetchWithTimeout(url, { method: "POST", headers: { @@ -232,6 +245,7 @@ Returns a permanent BlockRun-hosted URL.`, }, body: JSON.stringify(body), }, 95_000); + paidRequestInFlight = false; if (submitResp.status === 402) { throw new Error("Payment rejected. Check your wallet balance."); @@ -244,6 +258,7 @@ Returns a permanent BlockRun-hosted URL.`, let track: { url: string; duration_seconds?: number; lyrics?: string } | undefined; let modelReturned: string | undefined; let txHash: string | null | undefined; + let spendBooked = false; if (submitResp.status === 202) { // Async slow path: poll with the SAME payment header until completed. @@ -254,6 +269,7 @@ Returns a permanent BlockRun-hosted URL.`, // resolveGatewayUrl, not concatenation: it pins the poll to the same // origin that took the payment and refuses a cross-origin redirect. const pollAbsoluteUrl = resolveGatewayUrl(submitData.poll_url); + jobId = submitData.id; const startedAt = Date.now(); // Two independent deadlines, and the loop must respect BOTH. The poll @@ -278,10 +294,24 @@ Returns a permanent BlockRun-hosted URL.`, const pollTimeoutMs = pollTimeoutFor(deadline, Date.now(), MUSIC_POLL_TIMEOUT_MS); if (pollTimeoutMs === 0) break; - const pollResp = await fetchWithTimeout(pollAbsoluteUrl, { - method: "GET", - headers: { "PAYMENT-SIGNATURE": paymentPayload }, - }, pollTimeoutMs); + let pollResp: Response; + paidRequestInFlight = true; + try { + pollResp = await fetchWithTimeout(pollAbsoluteUrl, { + method: "GET", + headers: { "PAYMENT-SIGNATURE": paymentPayload }, + }, pollTimeoutMs); + } catch { + // Polling is idempotent and settlement has not been observed. A + // transient disconnect is safe to retry inside the existing + // deadline (the EIP-3009 nonce is single-use, so re-sending the + // same header after a lost-in-flight settlement cannot settle + // twice), and one reset must not abandon a paid job. + // paidRequestInFlight stays true: the request that never answered + // may still be settling server-side. + continue; + } + paidRequestInFlight = false; const pollData = await pollResp.json().catch(() => ({})) as { status?: string; @@ -291,6 +321,18 @@ Returns a permanent BlockRun-hosted URL.`, }; lastStatus = pollData.status || lastStatus; + // Settlement happens SERVER-SIDE on the first poll the gateway + // answers "completed" — the USDC is gone the moment we observe it, + // whatever the rest of the payload looks like. Book immediately: + // validating first meant a malformed completed body threw, the + // catch returned an error, and finally released the reservation — + // a real charge the ledger never saw (the fix video.ts got in + // 0.39.1, which music did not). + if (lastStatus === "completed" && !spendBooked) { + recordActualSpend(budget, quotedUsd, MUSIC_COST, agent_id); + spendBooked = true; + } + if (pollResp.status === 202 && (lastStatus === "queued" || lastStatus === "in_progress")) continue; if (lastStatus === "failed") throw new Error(`Upstream generation failed: ${pollData.error || "unknown"}. No payment taken.`); if (pollResp.ok && lastStatus === "completed") { @@ -306,30 +348,53 @@ Returns a permanent BlockRun-hosted URL.`, } // 504 on poll = transient upstream poll timeout — retry. } - if (!track) throw new Error(`Music generation did not complete within ${Math.round(MUSIC_POLL_BUDGET_MS / 1000)}s (last status: ${lastStatus}). No payment was taken.`); + if (!track) { + // Whether money moved depends on paidRequestInFlight, which the + // catch reads; the message here states only what was observed. + throw new Error(`Music generation did not complete within ${Math.round(MUSIC_POLL_BUDGET_MS / 1000)}s (last status: ${lastStatus}).`); + } } else { - // Inline fast path (200): settled inline. Read the receipt first and - // parse defensively — a truncated body must not un-record a charge that - // already settled on-chain. + // Inline fast path (200): a 200 on this route IS a settlement — the + // gateway settles on-chain before it answers. Book the charge NOW, + // before reading the body (speech.ts does the same): a truncated body + // or a stripped receipt header must not un-record money that moved. txHash = submitResp.headers.get("X-Payment-Receipt") || submitResp.headers.get("x-payment-receipt"); + recordActualSpend(budget, quotedUsd, MUSIC_COST, agent_id); + spendBooked = true; const data = await submitResp.json().catch(() => null) as { data?: Array<{ url: string; duration_seconds?: number; lyrics?: string }>; model?: string } | null; track = data?.data?.[0]; modelReturned = data?.model; - if (!track?.url) { - if (txHash) recordActualSpend(budget, amountToUsd(details.amount), MUSIC_COST, agent_id); - throw new Error("No track URL in response"); - } + if (!track?.url) throw new Error("No track URL in response"); } // Real settled price from the 402 quote; fall back to the flat estimate // if it didn't parse. Surfaced in the footer so the user always sees the // charge without relying on the plugin's announce-cost skill. - const billedUsd = amountToUsd(details.amount) ?? MUSIC_COST; - recordActualSpend(budget, amountToUsd(details.amount), MUSIC_COST, agent_id); + const billedUsd = quotedUsd ?? MUSIC_COST; + // Backstop only — every reachable path here has already booked at the + // moment settlement was observed. + if (!spendBooked) recordActualSpend(budget, quotedUsd, MUSIC_COST, agent_id); return musicResult(track, modelReturned || model, billedUsd, txHash, false); } catch (err) { const errMsg = err instanceof Error ? err.message : String(err); + // The account rail bills at SUBMIT. A failure after that — deadline, + // poll error, terminal failure — leaves a charge the ledger must carry + // (finally releases the reservation, so without this booking the cap + // silently rises by the track price), and the one thing the caller must + // not do is "try again": that submits and bills a second job. Checked + // before isTimeoutError, which matches the deadline message and would + // glue retry advice onto a note saying the job was billed. + if (err instanceof BilledJobError) { + recordActualSpend(budget, err.paidUsd, MUSIC_COST, agent_id); + const what = err.billing === "billed" + ? `Music generation did not return a track, but the job was billed to the BlockRun account when the gateway accepted it${err.jobId ? ` (job ${err.jobId})` : ""}.` + : `Music generation got no answer to its submit, so the job MAY have been accepted and billed to the BlockRun account.`; + return { + content: [{ type: "text", text: `${what} Check https://user.blockrun.ai/dashboard/activity before doing anything else — a new blockrun_music call starts and bills a second job.\nError: ${errMsg}` }], + isError: true, + }; + } // "Fund your wallet" is the wrong remedy on the account rail — there is // no wallet, and launchTopUp() would try to provision one to send a card // onramp to. apiKeyAsyncPost already returns the correct message for a @@ -341,8 +406,26 @@ Returns a permanent BlockRun-hosted URL.`, }; } if (isTimeoutError(err)) { + const reclaim = jobId ? ` The finished job stays claimable on the gateway for ~48h (job ${jobId}); re-running blockrun_music would start and charge a new job.` : ""; + if (paidRequestInFlight) { + // A submit can settle inline (200) and the gateway settles a + // completed poll regardless of whether we are still connected, so + // a paid request that never answered is not "no charge". Book the + // quote conservatively — over-counting a slow request that settled + // nothing is the documented trade-off; under-counting a real charge + // is not. + recordActualSpend(budget, quotedUsd, MUSIC_COST, agent_id); + return { + content: [{ type: "text", text: `Music generation timed out while a request carrying the payment signature was still in flight, so the gateway MAY have settled the charge after this client gave up — check blockrun_wallet action:"report" or the wallet's recent transactions before retrying.${reclaim}\nError: ${errMsg}` }], + isError: true, + }; + } + // On Base, settlement happens only on a response the gateway sends + // as settled; the last one was not, so nothing settled. The Solana + // helper describes its own money state in errMsg. + const base = !isApiKeyMode() && getChain() !== "solana"; return { - content: [{ type: "text", text: `Music generation timed out. This can happen during peak load — please try again.\nError: ${errMsg}` }], + content: [{ type: "text", text: `Music generation timed out.${base ? ` No payment was taken.${reclaim}` : ""}\nError: ${errMsg}` }], isError: true, }; } diff --git a/src/tools/phone.ts b/src/tools/phone.ts index 3764832..e7f4383 100644 --- a/src/tools/phone.ts +++ b/src/tools/phone.ts @@ -88,6 +88,21 @@ Voice call flow + voice preset details + full body shapes in the \`phone\` skill if (hasPathTraversal(cleanPath)) { return { content: [{ type: "text", text: formatError(`Invalid path '${path}'.`) }], isError: true }; } + // Pin the namespace. Every sibling passthrough concatenates onto a fixed + // prefix (/v1/surf/, /v1/modal/, ...); this tool's prefix is /v1/ itself, + // so without this check no traversal was needed to reach another tool's + // route: `modal/sandbox/create` (up to $192) ran at the $0.012 unknown + // reserve above — past any budget cap, with a confirm dialog quoting the + // wrong number. Classify the route the gateway will serve (decoded, + // lowercased, query dropped), not the string the caller typed. Kept + // AFTER hasPathTraversal so `phone/../modal/...` is still named for what + // it is, and BEFORE reserveBudget so a refusal books nothing. + if (!/^(phone|voice)\//.test(normalizeClassifyPath(cleanPath))) { + return { + content: [{ type: "text", text: formatError(`Invalid path '${path}': blockrun_phone only serves phone/* and voice/* routes.`) }], + isError: true, + }; + } const estimatedCost = estimatePhoneCost(cleanPath, body !== undefined); const gate = reserveBudget(budget, agent_id, estimatedCost); if (!gate.allowed) { diff --git a/src/tools/price.ts b/src/tools/price.ts index 5ede051..1ba65bc 100644 --- a/src/tools/price.ts +++ b/src/tools/price.ts @@ -1,8 +1,22 @@ // src/tools/price.ts // // Pyth-backed market data tool. Crypto, FX and commodity are fully free -// (price + history + list); stocks (`stocks/{market}` and the `usstock` -// legacy alias) charge $0.001 per price or history call. +// (price + history + list). +// +// Equity is catalog-only. Since 2026-09-05 the gateway answers every +// `stocks/{market}/price` and `/history` call (and the `usstock` alias) with a +// pre-payment 501 — "We do not currently serve equity prices" — after +// blockrun#517 moved the free tier onto licensed sources. `stocks/{market}/list` +// still serves the ticker catalog for free (the Solana gateway has no equity +// route at all — it answers with the site HTML). +// +// Paid stock price/history is therefore answered HERE, before the chain guard +// and before any network call: on the default Solana chain the Base-only guard +// used to fire first and tell the user to switch chains to pay for a route +// that 501s. The gateway's answer is a product decision with a contact +// address, not a transient fault, so pre-empting it loses nothing. Retire the +// pre-flight (and re-enable the paid path below it) once +// `curl https://blockrun.ai/v1/stocks/us/price/AAPL` answers 402 again. // // Supported markets: us, hk, jp, kr, gb, de, fr, nl, ie, lu, cn, ca. @@ -18,7 +32,7 @@ import { reserveBudget, recordSpending } from "../utils/budget.js"; import { confirmSpend } from "../utils/confirm-spend.js"; import { withTxFee } from "../utils/tx-fee.js"; import type { BudgetState } from "../types.js"; -import { baseOnlyMessage, getPriceClient } from "../utils/wallet.js"; +import { getPriceClient } from "../utils/wallet.js"; import { extractErrorMessage, formatError } from "../utils/errors.js"; import { TOOL_ANNOTATIONS } from "../tool-annotations.js"; @@ -35,11 +49,25 @@ function isPaidPriceCall(action: "price" | "history" | "list", category: string) return action !== "list" && (category === "stocks" || category === "usstock"); } +/** + * What the gateway itself answers for equity price/history since 2026-09-05 + * (HTTP 501, verified live 2026-09-08), said before the wallet is consulted. + * Exported for the test; nothing here touches the network. + */ +export function equityNotServedMessage(action: string, category: string, market?: string): string { + const mkt = market ?? "us"; + return `Error: Equity ${action === "history" ? "history" : "quotes"} are not served (gateway 501 for category "${category}").\n\n` + + `The gateway withdrew equity price and history on 2026-09-05 — this is not an outage, retrying will not help, ` + + `and nothing was charged (the wallet was not asked to sign).\n` + + `The ticker catalog still works and is free: { action: "list", category: "stocks", market: "${mkt}" }.\n` + + `For realtime or global equity coverage, contact hello@blockrun.ai.`; +} + export function registerPriceTool(server: McpServer, budget: BudgetState): void { server.registerTool( "blockrun_price", { - description: `Realtime quotes and OHLC history for crypto, FX, commodities and 12 global stock markets (Pyth-backed). + description: `Realtime quotes and OHLC history for crypto, FX and commodities (Pyth-backed), plus the ticker catalog for 12 stock markets. - action="price" — realtime quote for a symbol - action="history" — OHLC bars between from/to (unix seconds) @@ -47,20 +75,20 @@ export function registerPriceTool(server: McpServer, budget: BudgetState): void Pricing: - crypto / fx / commodity: FREE across price, history and list -- stocks / usstock: $0.001 per price or history call (list free) +- stocks / usstock: list (ticker catalog) FREE; price/history NOT SERVED — gateway 501 before payment since 2026-09-05, nothing charged, do not retry Stocks markets: us, hk, jp, kr, gb, de, fr, nl, ie, lu, cn, ca (required when category="stocks"). Examples: - { action: "price", category: "crypto", symbol: "BTC-USD" } -- { action: "price", category: "stocks", symbol: "AAPL", market: "us" } +- { action: "price", category: "fx", symbol: "EUR-USD" } - { action: "history", category: "crypto", symbol: "ETH-USD", resolution: "D", from: 1700000000, to: 1710000000 } - { action: "list", category: "crypto", query: "sol" }`, annotations: TOOL_ANNOTATIONS.readOnlyOpenWorld, inputSchema: { action: ACTION.describe("Which endpoint to hit: price, history, or list."), category: CATEGORY.describe("Market category."), - symbol: z.string().optional().describe("Ticker (required for price+history). e.g. BTC-USD, AAPL, EUR-USD."), + symbol: z.string().optional().describe("Ticker (required for price+history). e.g. BTC-USD, EUR-USD, XAU-USD."), market: MARKET.optional().describe("Stock market code — required when category='stocks'."), session: SESSION.optional().describe("Equity session hint (pre/post/on); ignored for non-equity."), resolution: RESOLUTION.optional().describe("Bar resolution for history (default D)."), @@ -73,16 +101,23 @@ Examples: }, async ({ action, category, symbol, market, session, resolution, from, to, query, limit, agent_id }) => { try { + // Equity price/history first — before the market-required throw, so the + // most natural stocks call (no market) gets the real answer in one round + // trip instead of a validation error for a route that is not served. + const paid = isPaidPriceCall(action, category); + if (paid) { + return { + content: [{ type: "text", text: equityNotServedMessage(action, category, market) }], + isError: true, + }; + } + // Re-enable when the equity route returns (see the header): the paid + // path is Base-only, so restore `baseOnlyMessage("Paid stock price/history calls")` + // here ahead of the reservation. if (category === "stocks" && !market) { throw new Error("market is required when category='stocks'"); } - const paid = isPaidPriceCall(action, category); - const chainBlock = paid ? baseOnlyMessage("Paid stock price/history calls") : null; - if (chainBlock) { - return { content: [{ type: "text", text: formatError(chainBlock) }], isError: true }; - } - // withTxFee: the gateway charges base + $0.002 (src/utils/tx-fee.ts), so a // paid stock call settles at $0.0030 against a $0.001 base — reserving the // base was 3x short. Free categories (crypto/fx/commodity) stay $0. diff --git a/src/tools/realface.ts b/src/tools/realface.ts index 0a6a3b2..4206565 100644 --- a/src/tools/realface.ts +++ b/src/tools/realface.ts @@ -337,11 +337,16 @@ Privacy: BlockRun does not store face/liveness data — only the asset id, name, throw new Error(`Portrait enroll error ${status}: ${data.error || JSON.stringify(data)}`); } + // The gateway answers 2xx only AFTER settling, so the charge is real + // whatever the body looks like. Book it before validating the payload: + // a truncated or asset-less body used to throw first, the catch + // formatted a failure, and finally released the reservation — a real + // charge the ledger never saw (same ordering video.ts and speech.ts fixed). + recordActualSpend(budget, settledUsd, ENROLLMENT_PRICE_USD, agent_id); + const assetId: string | undefined = data.asset_id; if (!assetId) throw new Error(`Portrait response missing asset_id: ${JSON.stringify(data)}`); - recordActualSpend(budget, settledUsd, ENROLLMENT_PRICE_USD, agent_id); - const txHash = data.settlement?.tx_hash || undefined; const lines = [ `✅ Virtual Portrait enrolled!`, @@ -406,11 +411,13 @@ Privacy: BlockRun does not store face/liveness data — only the asset id, name, throw new Error(`Enroll error ${status}: ${data.error || JSON.stringify(data)}`); } + // Book before validating the payload — see the portrait action above: + // a settled 2xx with a malformed body must not un-record the charge. + recordActualSpend(budget, settledUsd, ENROLLMENT_PRICE_USD, agent_id); + const assetId: string | undefined = data.asset_id; if (!assetId) throw new Error(`Enroll response missing asset_id: ${JSON.stringify(data)}`); - recordActualSpend(budget, settledUsd, ENROLLMENT_PRICE_USD, agent_id); - const txHash = data.settlement?.tx_hash || undefined; const lines = [ `✅ RealFace enrolled!`, diff --git a/src/tools/rpc.ts b/src/tools/rpc.ts index 7be2b81..92509ab 100644 --- a/src/tools/rpc.ts +++ b/src/tools/rpc.ts @@ -27,7 +27,7 @@ export function registerRpcTool(server: McpServer, budget: BudgetState): void { server.registerTool( "blockrun_rpc", { - description: `Raw JSON-RPC against 40+ blockchains — one endpoint, no node, no API key. $0.002 per call (batch charges per element). + description: `Raw JSON-RPC against 40+ blockchains — one endpoint, no node, no API key. $0.002 base per call plus the gateway's flat tx fee ($0.001 today; we reserve $0.002, so budget/confirm show $0.004 per single call). A JSON-RPC batch charges $0.002 per element plus ONE fee — batch when you can. Use when you need data the higher-level tools don't cover: contract reads (eth_call), balances, blocks, txs, logs, gas, or any chain-native RPC method. @@ -40,7 +40,7 @@ Examples: blockrun_rpc({ network: "bitcoin", method: "getblockchaininfo" }) blockrun_rpc({ network: "ethereum", body: [{jsonrpc:"2.0",id:1,method:"eth_blockNumber"},{...}] }) // batch -Prefer blockrun_price (free quotes), blockrun_dex (free DEX data), or blockrun_surf (labeled/aggregated data) when they cover the question — this tool is for raw chain access.`, +Prefer blockrun_price (free quotes) or blockrun_dex (free DEX data) when they cover the question — this tool is for raw chain access.`, annotations: TOOL_ANNOTATIONS.publicOrExternalWrite, inputSchema: { network: z.string().describe("Chain key, e.g. 'ethereum', 'base', 'solana', 'bitcoin', 'arbitrum', 'polygon'. Unknown slugs pass through to the Tatum gateway."), diff --git a/src/tools/surf.ts b/src/tools/surf.ts deleted file mode 100644 index 51b5d60..0000000 --- a/src/tools/surf.ts +++ /dev/null @@ -1,129 +0,0 @@ -// src/tools/surf.ts -// -// Surf (asksurf.ai) — unified crypto data API. Path-based passthrough so the -// 83-endpoint catalog stays out of the tool description (it lives in the surf -// skill instead). Adding new Surf endpoints does not require an MCP release. -// -// Mirrors the markets.ts pattern. Method is inferred: pass `body` for POST -// (onchain/query, onchain/sql), otherwise GET with `params`. -// -// Settlement: each call settles directly to Surf's Base treasury. BlockRun -// forwards the request server-side using the BlockRun-held SURF_API_KEY. - -import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; -import { TOOL_ANNOTATIONS } from "../tool-annotations.js"; -import { z } from "zod"; -import { reserveBudget, recordSpending, recordActualSpend } from "../utils/budget.js"; -import { confirmSpend } from "../utils/confirm-spend.js"; -import { asStructuredContent, coerceBody } from "../utils/body.js"; -import { getClient } from "../utils/wallet.js"; -import { type RawClient, rawPost, rawGet } from "../utils/raw-call.js"; -import { formatError, extractErrorMessage } from "../utils/errors.js"; -import { hasPathTraversal } from "../utils/path-safety.js"; -import type { BudgetState } from "../types.js"; - -type SurfClient = { - getWithPaymentRaw: (endpoint: string, params?: Record) => Promise; - requestWithPaymentRaw: (endpoint: string, body: unknown) => Promise; -}; - -// Flat per-call price CHARGED for every Surf endpoint: $0.0075 base + $0.002 -// flat transaction fee. Keep in step with SURF_TIER_*_PRICE in the gateway's -// src/lib/surf.ts, and note that constant is the BASE — not what a caller pays. -export const SURF_PRICE_USD = 0.0095; - -// Exported for unit tests. -// -// Surf is a FLAT $0.0095/call — every endpoint, every former tier (gateway change -// 2026-07-15: one network-uniform price across Surf and Predexon). The old T1/T2/T3 -// tier sets are gone: they no longer affect price, and keeping them here only -// invited the reader to believe otherwise. Verified live across every tier — -// market/price, wallet/detail and onchain/sql all quote 9500 micro. -// -// This estimator feeds the BUDGET GATE, so it must never under-quote — and the -// number to quote is what x402 CHARGES, not the 402's JSON `price` field. That -// field reports the base ($0.0075); the charge is in `maxAmountRequired` inside -// the base64 `payment-required` header, and every /v1/surf/* route decodes to -// 9500 micro = $0.0095 (verified live 2026-07-15). -// -// This has now been wrong twice in the same direction, both times by trusting a -// number that looked authoritative: first the stale $0.001/$0.005/$0.02 tiers -// after the gateway went flat, then the $0.0075 base after it was mistaken for -// the price. Read the header. -export function estimateSurfCost(_path: string): number { - return SURF_PRICE_USD; -} - -export function registerSurfTool(server: McpServer, budget: BudgetState): void { - server.registerTool( - "blockrun_surf", - { - description: `Unified crypto data via Surf (asksurf.ai) — 83 endpoints, one API. - -Coverage: CEX market data (16 exchanges), on-chain SQL across 13 chains, 100M+ labeled wallets, prediction markets (Polymarket + Kalshi), social mindshare / CT intelligence, news, and unified search. - -Pricing (settled in USDC to Surf's Base treasury): -- Flat $0.0095/call — every endpoint, including raw on-chain SQL. No tiers. ($0.0075 base + $0.002 tx fee.) - -Common paths (full 83-endpoint catalog in the surf skill): -- market/price?symbol=BTC -- exchange/price?pair=BTC-USDT -- prediction-market/polymarket/ranking -- search/web?q=ethereum+pectra+upgrade -- wallet/detail?address=0x... -- social/mindshare?q=ethereum&interval=1d -- onchain/sql + body:{ sql: "SELECT ..." } - -Method is auto-routed: pass 'body' for POST endpoints; otherwise GET with 'params'. -Each Surf endpoint pre-validates required params before settling — you get a 400 (not a charge) if a required field is missing. Browse the full catalog: https://blockrun.ai/marketplace/surf`, - annotations: TOOL_ANNOTATIONS.readOnlyOpenWorld, - inputSchema: { - path: z.string().describe("Endpoint path under /v1/surf/, e.g. 'market/price', 'prediction-market/polymarket/ranking', 'wallet/detail', 'onchain/sql'"), - params: z.record(z.string(), z.string()).optional().describe("Query parameters for GET endpoints, e.g. { symbol: 'BTC' } or { address: '0x...', chain: 'ethereum' }"), - body: z.any().optional().describe("JSON body for POST endpoints. Provide for: onchain/query, onchain/sql. When set, the call is sent as POST; otherwise GET with params."), - agent_id: z.string().optional().describe("Agent identifier for budget tracking and enforcement."), - }, - }, - async ({ path, params, body, agent_id }) => { - try { - body = coerceBody(body); - const cleanPath = path.replace(/^\/+/, "").replace(/^v1\/surf\//, "").replace(/^api\/v1\/surf\//, ""); - if (hasPathTraversal(cleanPath)) { - return { content: [{ type: "text", text: formatError(`Invalid path '${path}'.`) }], isError: true }; - } - const estimatedCost = estimateSurfCost(cleanPath); - const gate = reserveBudget(budget, agent_id, estimatedCost); - if (!gate.allowed) { - return { - content: [{ type: "text", text: `${gate.reason}. Use blockrun_wallet action:"report" to see usage or action:"delegate" to increase agent budget.` }], - isError: true, - }; - } - try { - // Human-in-the-loop (BLOCKRUN_CONFIRM_SPEND=on): ask before signing. A - // decline returns here — nothing is sent, and the finally releases the - // reservation. No-ops when off, sub-threshold, or unsupported by the client. - const confirm = await confirmSpend(server, { usd: estimatedCost, label: `surf · ${cleanPath}` }); - if (!confirm.ok) return { content: [{ type: "text", text: confirm.reason ?? "Charge cancelled." }] }; - const client = getClient() as unknown as SurfClient; - const endpoint = `/v1/surf/${cleanPath}`; - const { data: result, paidUsd } = body !== undefined - ? await rawPost(client, endpoint, body) - : await rawGet(client, endpoint, params); - recordActualSpend(budget, paidUsd, estimatedCost, agent_id); - return { - content: [{ type: "text", text: JSON.stringify(result, null, 2) }], - structuredContent: asStructuredContent(result), - }; - } finally { - gate.release(); - } - } catch (err) { - return { - content: [{ type: "text", text: formatError(extractErrorMessage(err)) }], - isError: true, - }; - } - } - ); -} diff --git a/src/tools/video.ts b/src/tools/video.ts index 82e5c25..d24b26c 100644 --- a/src/tools/video.ts +++ b/src/tools/video.ts @@ -2,7 +2,7 @@ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; import { TOOL_ANNOTATIONS } from "../tool-annotations.js"; import { z } from "zod"; -import { amountToUsd, reserveBudget, recordActualSpend } from "../utils/budget.js"; +import { amountToUsd, assertQuoteNearEstimate, reserveBudget, recordActualSpend } from "../utils/budget.js"; import { confirmSpend } from "../utils/confirm-spend.js"; import { withTxFee } from "../utils/tx-fee.js"; import { formatError, isPaymentRejectionError } from "../utils/errors.js"; @@ -12,7 +12,7 @@ import { pollTimeoutFor } from "../utils/poll.js"; import type { BudgetState } from "../types.js"; import { getApiBase, getChain, getOrCreateWalletKey, resolveGatewayUrl } from "../utils/wallet.js"; import { isApiKeyMode } from "../utils/auth.js"; -import { apiKeyAsyncPost } from "../utils/api-key-call.js"; +import { apiKeyAsyncPost, BilledJobError } from "../utils/api-key-call.js"; import { isBlockedFetchHostResolved } from "../utils/ssrf.js"; import { privateKeyToAccount } from "viem/accounts"; import { @@ -274,16 +274,38 @@ export function estimateVideoCost(model: string, durationSeconds?: number, resol return withTxFee(VIDEO_BASE_PRICE_PER_SECOND[model] * seconds * VIDEO_MARGIN); } +/** + * Refuse a 402 whose price is far above what the estimator (and the description + * the model read) said this call costs. See assertQuoteNearEstimate for the + * rule; this adds the one hint that is video-specific today. Verified live + * 2026-09-08: sol.blockrun.ai quotes azure/sora-2 as "Seedance 2.0 Pro video + * generation (5s)" at $1.135480 against Base's $0.421001. + */ +export function assertVideoQuoteSane( + quotedUsd: number | null, + estimatedCost: number, + model: string, + chain: "base" | "solana", + quotedFor?: string, +): void { + const hint = chain === "solana" + ? (model === "azure/sora-2" + ? `The Solana gateway is a separate deployment and does not serve azure/sora-2 yet — it quotes Seedance 2.0 in its place. Switch to Base for Sora (blockrun_wallet action:"chain" chain:"base") or pick a Seedance model explicitly.` + : `Retry on Base (blockrun_wallet action:"chain" chain:"base") or pick another model.`) + : `Retry on Solana (blockrun_wallet action:"chain" chain:"solana") or pick another model.`; + assertQuoteNearEstimate(quotedUsd, estimatedCost, { what: `${model} video`, quotedFor, hint }); +} + export function registerVideoTool(server: McpServer, budget: BudgetState): void { server.registerTool( "blockrun_video", { description: `Generate short AI videos via BlockRun x402 on the active Base or Solana chain (async, client-polled). -Turns a text prompt (and optional seed image) into a short MP4 clip. The tool submits the job, then polls until the video is ready (typical total wall-time 60-180s; 9 min Base / 15 min Solana hard cap). Payment is settled only when upstream returns a finished video — if the job fails or we give up, you are not charged. +Turns a text prompt (and optional seed image) into a short MP4 clip. The tool submits the job, then polls until the video is ready (typical total wall-time 60-180s; 9 min Base / 15 min Solana hard cap). Payment is settled only when upstream returns a finished video — if the job fails you are not charged; if this client gives up while a paid poll is still in flight the gateway may still settle, and the error text says so. Models. Every rate below is what you are CHARGED (margin and transaction fee included), at the 720p baseline Seedance renders by default with synced audio: -- azure/sora-2 (~$0.105/sec, 720p + synced audio, text-to-video) — OpenAI Sora 2 via Azure AI Foundry. duration_seconds must be 4, 8, or 12 (4s default -> ~$0.42/clip). No image_url / RealFace. +- azure/sora-2 (~$0.105/sec, 720p + synced audio, text-to-video) — OpenAI Sora 2 via Azure AI Foundry. duration_seconds must be 4, 8, or 12 (4s default -> ~$0.42/clip). No image_url / RealFace. Base only for now: the Solana gateway quotes it as Seedance 2.0 at $1.135 and the tool refuses that quote unsigned. - xai/grok-imagine-video ($0.05/sec at 480p default, $0.07/sec at 720p; 8s default -> $0.401/clip, 1-15s) — stylized, fast. 480p/720p only. - bytedance/seedance-1.5-pro (~$0.071/sec, 4-12s, 5s default -> ~$0.35/clip) — cheapest Seedance, token-priced upstream - bytedance/seedance-2.0-mini (~$0.080/sec, 4-15s, 5s default) — 2.0-generation quality at roughly half the 2.0-fast rate; 720p ceiling; supports RealFace and first/last-frame @@ -314,6 +336,16 @@ Returns a permanent blockrun-hosted MP4 URL (the gateway mirrors the asset to GC // Reserve the estimate up front so concurrent calls can't each pass a // stale budget; release in finally once the call settles or fails. let gate: ReturnType | undefined; + // Visible to the catch, which has to book money that moved without a + // result: the account rail bills at submit, and a Base poll aborted in + // flight can still settle server-side. Every give-up also names the job. + let estimatedCost = 0; + let quotedUsd: number | null = null; + let jobId: string | undefined; + // True while a Base poll carrying the payment header has been issued and + // has not answered. A poll that rejects leaves it true: that request may + // still be settling on the gateway, which does not stop on disconnect. + let paidPollInFlight = false; try { const selectedModel = model || "xai/grok-imagine-video"; @@ -418,7 +450,7 @@ Returns a permanent blockrun-hosted MP4 URL (the gateway mirrors the asset to GC // Image input is NOT discounted upstream on Seedance (only video-to-video // is), so text-to-video and image-to-video share one per-second rate. - const estimatedCost = estimateVideoCost(selectedModel, billedSeconds, resolution); + estimatedCost = estimateVideoCost(selectedModel, billedSeconds, resolution); gate = reserveBudget(budget, agent_id, estimatedCost); if (!gate.allowed) { return { @@ -497,7 +529,10 @@ Returns a permanent blockrun-hosted MP4 URL (the gateway mirrors the asset to GC body, { pollBudgetMs: SOLANA_VIDEO_TOTAL_BUDGET_MS, - onQuote: (quotedUsd) => { + onQuote: (quotedUsd, quoteDetails) => { + // WHAT was quoted, before how much: a substituted or repriced + // model is refused here, unsigned (QuoteMismatchError). + assertVideoQuoteSane(quotedUsd, estimatedCost, selectedModel, "solana", quoteDetails?.resource?.description); if (quotedUsd === null || quotedUsd <= estimatedCost) return; gate?.release(); gate = reserveBudget(budget, agent_id, quotedUsd); @@ -578,6 +613,16 @@ Returns a permanent blockrun-hosted MP4 URL (the gateway mirrors the asset to GC isError: true, }; } + quotedUsd = settledUsd; + + // WHAT was quoted, before how much. The estimator tracks the live 402 to + // within a cent (verify:prices), so a quote far above it is a reprice or + // a substituted model — refuse it unsigned rather than re-reserve it. + try { + assertVideoQuoteSane(settledUsd, estimatedCost, selectedModel, "base", details.resource?.description); + } catch (err) { + return { content: [{ type: "text", text: formatError(err instanceof Error ? err.message : String(err)) }], isError: true }; + } // The 402 carries the REAL price; Seedance/Sora are token-priced, so a // 1080p/4K render can far exceed the per-second estimate reserved at @@ -648,6 +693,7 @@ Returns a permanent blockrun-hosted MP4 URL (the gateway mirrors the asset to GC // that took the payment. A paid job polled on the wrong host is money // spent for a result that can never be collected. const pollAbsoluteUrl = resolveGatewayUrl(submitData.poll_url); + jobId = submitData.id; let lastStatus = submitData.status || "queued"; let spendBooked = false; @@ -673,10 +719,24 @@ Returns a permanent blockrun-hosted MP4 URL (the gateway mirrors the asset to GC const pollTimeoutMs = pollTimeoutFor(deadline, Date.now(), VIDEO_POLL_TIMEOUT_MS); if (pollTimeoutMs === 0) break; - const pollResp = await fetchWithTimeout(pollAbsoluteUrl, { - method: "GET", - headers: { "PAYMENT-SIGNATURE": paymentPayload }, - }, pollTimeoutMs); + let pollResp: Response; + paidPollInFlight = true; + try { + pollResp = await fetchWithTimeout(pollAbsoluteUrl, { + method: "GET", + headers: { "PAYMENT-SIGNATURE": paymentPayload }, + }, pollTimeoutMs); + } catch { + // Polling is idempotent and settlement has not been observed. A + // transient disconnect is safe to retry inside the existing + // deadline (the EIP-3009 nonce is single-use, so re-sending the + // same header after a lost-in-flight settlement cannot settle + // twice), and one reset must not abandon a nine-minute render. + // paidPollInFlight stays true: the request that never answered may + // still be settling server-side. + continue; + } + paidPollInFlight = false; const pollData = await pollResp.json().catch(() => ({})) as { status?: string; @@ -736,7 +796,9 @@ Returns a permanent blockrun-hosted MP4 URL (the gateway mirrors the asset to GC } if (!completed) { - throw new Error(`Video generation did not complete within ${Math.round(VIDEO_TOTAL_BUDGET_MS / 1000)}s (last status: ${lastStatus}). No payment was taken.`); + // Whether money moved depends on paidPollInFlight, which the catch + // reads; the message here states only what was observed. + throw new Error(`Video generation did not complete within ${Math.round(VIDEO_TOTAL_BUDGET_MS / 1000)}s (last status: ${lastStatus}).`); } // Real settled price from the 402 (token-priced upstream); fall back to @@ -772,6 +834,23 @@ Returns a permanent blockrun-hosted MP4 URL (the gateway mirrors the asset to GC }; } catch (err) { const errMsg = err instanceof Error ? err.message : String(err); + // The account rail bills at SUBMIT. A failure after that — deadline, + // poll error, terminal failure — leaves a charge the ledger must carry + // (finally releases the reservation, so without this booking the cap + // silently rises by the clip price), and the one thing the caller must + // not do is "try again": that submits and bills a second job. Checked + // before isTimeoutError, which matches the deadline message and would + // glue retry advice onto a note saying the job was billed. + if (err instanceof BilledJobError) { + recordActualSpend(budget, err.paidUsd, estimatedCost, agent_id); + const what = err.billing === "billed" + ? `Video generation did not return a clip, but the job was billed to the BlockRun account when the gateway accepted it${err.jobId ? ` (job ${err.jobId})` : ""}.` + : `Video generation got no answer to its submit, so the job MAY have been accepted and billed to the BlockRun account.`; + return { + content: [{ type: "text", text: `${what} Check https://user.blockrun.ai/dashboard/activity before doing anything else — a new blockrun_video call starts and bills a second job.\nError: ${errMsg}` }], + isError: true, + }; + } if (isPaymentRejectionError(errMsg)) { return { content: [{ type: "text", text: `Video generation needs USDC — your wallet is out of funds. ${(await launchTopUp()).note}\nError: ${errMsg}` }], @@ -779,8 +858,25 @@ Returns a permanent blockrun-hosted MP4 URL (the gateway mirrors the asset to GC }; } if (isTimeoutError(err)) { + const reclaim = jobId ? ` The finished job stays claimable on the gateway for ~48h (job ${jobId}); re-running blockrun_video would start and charge a new job.` : ""; + if (paidPollInFlight) { + // The gateway's poll route passes the request signal only to the + // upstream check; on "completed" it backs up the clip and settles + // regardless of whether we are still connected. Book the quote + // conservatively — over-counting a slow poll that settled nothing + // is the documented trade-off; under-counting a real charge is not. + recordActualSpend(budget, quotedUsd, estimatedCost, agent_id); + return { + content: [{ type: "text", text: `Video generation timed out while a poll carrying the payment signature was still in flight, so the gateway MAY have settled the charge after this client gave up — check blockrun_wallet action:"report" or the wallet's recent transactions before retrying.${reclaim}\nError: ${errMsg}` }], + isError: true, + }; + } + // On Base, settlement happens only on a poll the gateway answers + // "completed"; the last one answered otherwise, so nothing settled. + // The Solana helper describes its own money state in errMsg. + const base = !isApiKeyMode() && getChain() !== "solana"; return { - content: [{ type: "text", text: `Video generation timed out. The upstream async job didn't complete in time — please try again.\nError: ${errMsg}` }], + content: [{ type: "text", text: `Video generation timed out.${base ? ` No payment was taken.${reclaim}` : ""}\nError: ${errMsg}` }], isError: true, }; } diff --git a/src/utils/api-key-call.ts b/src/utils/api-key-call.ts index ba31ddc..710ff9f 100644 --- a/src/utils/api-key-call.ts +++ b/src/utils/api-key-call.ts @@ -75,6 +75,72 @@ function costFrom(response: Response): number | null { /** Statuses the gateway uses for a job that will never complete. */ const TERMINAL_FAILURES = new Set(["failed", "cancelled", "canceled"]); +/** + * Poll statuses that mean "the proxy tier hiccupped", not "the job is gone". + * The SDK's own ApiKeyAuth.fetch retries 502/503/504/522/524 on GETs; 429 is + * added because a poll is free and Retry-After says exactly how long to wait. + * On this rail the job is already paid for, so abandoning it on one of these + * costs the whole clip and invites a second, equally billed submit. + */ +const TRANSIENT_POLL_STATUSES = new Set([429, 502, 503, 504, 522, 524]); + +/** + * A failure on the account rail's async path for which the account has been, or + * may have been, charged. + * + * This rail bills a media job the moment the gateway accepts it (202); the polls + * are free. So unlike the wallet rails, where "we gave up" means "nothing + * settled", every exit after a successful submit here leaves money spent — and + * the two things a caller needs are exactly what a bare Error cannot carry: how + * much (to book it), and which job (so nobody submits it twice). The message text + * of the deadline case is unchanged from before this class existed, because + * isTimeoutError keys on it. + */ +export class BilledJobError extends Error { + /** + * The settled cost from the submit response's x-blockrun-cost-usd, or null + * when the header was absent. Null is not free: callers hand it to + * recordActualSpend, which falls back to their estimate, never to $0. + */ + readonly paidUsd: number | null; + readonly jobId?: string; + /** + * "billed": the gateway answered 202, so the charge is certain. + * "unknown": no response was observed — the submit never returned, or a + * terminal failure arrived without a payment_status — so the request may or + * may not have been billed. Callers book the estimate in both cases: a cap + * that over-counts a lost request is the safe direction, and under-counting a + * real charge is the failure the ledger exists to prevent. + */ + readonly billing: "billed" | "unknown"; + + constructor(message: string, opts: { paidUsd: number | null; jobId?: string; billing: "billed" | "unknown" }) { + super(message); + this.name = "BilledJobError"; + this.paidUsd = opts.paidUsd; + this.jobId = opts.jobId; + this.billing = opts.billing; + } +} + +/** + * True when a fetch rejection proves the request never left this machine — DNS + * failed, or the connection was refused — so nothing could have been billed. + * Anything else (an abort, a reset, a socket error mid-flight) is ambiguous: + * the request may have reached the gateway and been accepted. + */ +const NEVER_CONNECTED = new Set(["ENOTFOUND", "EAI_AGAIN", "ECONNREFUSED", "ENETUNREACH", "EHOSTUNREACH", "EADDRNOTAVAIL"]); +function connectionNeverOpened(err: unknown): boolean { + const cause = (err as { cause?: { code?: unknown } } | undefined)?.cause; + return typeof cause?.code === "string" && NEVER_CONNECTED.has(cause.code); +} + +function retryAfterMs(response: Response): number { + const raw = response.headers.get("retry-after"); + const seconds = raw === null ? NaN : Number(raw.trim()); + return Number.isFinite(seconds) && seconds > 0 ? seconds * 1000 : 0; +} + function receiptFrom(response: Response): string | undefined { return ( response.headers.get("x-payment-receipt") ?? @@ -87,21 +153,21 @@ async function readJson(response: Response): Promise> { return (await response.json().catch(() => ({}))) as Record; } -async function throwForStatus(response: Response, what: string): Promise { - const body = await readJson(response); +/** The message for a non-ok response whose body has already been read. */ +function statusErrorMessage(response: Response, what: string, body: Record): string { // A 402 on this rail is not a quote to pay — it means the ACCOUNT is out of // credit. Signing anything here would be wrong (there is no wallet), so say // what actually has to happen. if (response.status === 402) { - throw new Error( + return ( `${what} was refused: the BlockRun account is out of credit. ` + - `Top up at https://user.blockrun.ai/dashboard/credits.`, + `Top up at https://user.blockrun.ai/dashboard/credits.` ); } if (response.status === 401) { - throw new Error( + return ( `${what} was refused: the BlockRun API key was rejected. ` + - `Check the key at https://user.blockrun.ai/dashboard/keys.`, + `Check the key at https://user.blockrun.ai/dashboard/keys.` ); } // Surface Retry-After rather than burying it in the body. It is the one piece @@ -110,11 +176,13 @@ async function throwForStatus(response: Response, what: string): Promise // PR #136, which surfaced it and this path did not.) if (response.status === 429) { const retry = response.headers.get("retry-after"); - throw new Error( - `${what} was rate limited${retry ? ` — retry after ${retry}s` : ""}.`, - ); + return `${what} was rate limited${retry ? ` — retry after ${retry}s` : ""}.`; } - throw new Error(`API error ${response.status}: ${JSON.stringify(body)}`); + return `API error ${response.status}: ${JSON.stringify(body)}`; +} + +async function throwForStatus(response: Response, what: string): Promise { + throw new Error(statusErrorMessage(response, what, await readJson(response))); } /** POST an endpoint that answers inline. `endpoint` is rooted, e.g. "/v1/audio/speech". */ @@ -188,15 +256,32 @@ export async function apiKeyAsyncPost( // only deadline is the caller's own budget. const deadline = startedAt + pollBudgetMs; - const submit = await fetchWithTimeout( - `${getApiBase()}${endpoint}`, - { - method: "POST", - headers: { "Content-Type": "application/json", ...apiAuthHeaders() }, - body: JSON.stringify(body), - }, - opts.submitTimeoutMs ?? 95_000, - ); + let submit: Response; + try { + submit = await fetchWithTimeout( + `${getApiBase()}${endpoint}`, + { + method: "POST", + headers: { "Content-Type": "application/json", ...apiAuthHeaders() }, + body: JSON.stringify(body), + }, + opts.submitTimeoutMs ?? 95_000, + ); + } catch (err) { + // A submit that never connected cannot have been billed; let it surface as + // the network error it is. Anything else is ambiguous — the request may + // have reached the gateway, which bills the moment it accepts — and the + // honest statement is "may have", not "was" (no charge was observed) and + // not "was not" (which would license a second submit). + if (connectionNeverOpened(err)) throw err; + const reason = err instanceof Error ? err.message : String(err); + throw new BilledJobError( + `POST ${endpoint} did not return a response (${reason}). The request may have reached the gateway, ` + + `and this rail bills a job the moment it is accepted, so the job MAY have been accepted and billed to the account — ` + + `check https://user.blockrun.ai/dashboard/activity before submitting again.`, + { paidUsd: null, billing: "unknown" }, + ); + } if (!submit.ok && submit.status !== 202) await throwForStatus(submit, `POST ${endpoint}`); const submitted = await readJson(submit); @@ -214,16 +299,32 @@ export async function apiKeyAsyncPost( const absolutePollUrl = resolveGatewayUrl(pollUrl); let lastStatus = typeof submitted.status === "string" ? submitted.status : "queued"; + // From here on the account has paid. Every give-up below says so, names the + // job, and carries the cost, because the natural next move after a bare + // failure is to submit again — and that bills a second job. + const billedNote = + `It has already been billed to the account${jobId ? `; job id ${jobId}` : ""} — ` + + `check https://user.blockrun.ai/dashboard/activity before submitting again.`; + const billed = (message: string) => new BilledJobError(message, { paidUsd: submitCost, jobId, billing: "billed" }); + while (Date.now() < deadline) { await new Promise((r) => setTimeout(r, pollIntervalMs)); const timeout = pollTimeoutFor(deadline, Date.now(), pollTimeoutMs); if (timeout === 0) break; - const poll = await fetchWithTimeout( - absolutePollUrl, - { method: "GET", headers: { ...apiAuthHeaders() } }, - timeout, - ); + let poll: Response; + try { + poll = await fetchWithTimeout( + absolutePollUrl, + { method: "GET", headers: { ...apiAuthHeaders() } }, + timeout, + ); + } catch { + // Polls are free and idempotent, and the money is already gone: a + // transient disconnect (or one clamped poll's abort) must not abandon a + // job the account has paid for. The deadline above bounds the retry. + continue; + } const data = await readJson(poll); if (typeof data.status === "string") lastStatus = data.status; @@ -236,12 +337,17 @@ export async function apiKeyAsyncPost( // it tells someone not to check a charge that may be real. const paymentStatus = typeof data.payment_status === "string" ? data.payment_status : undefined; const note = typeof data.note === "string" ? data.note : undefined; + const failed = `Upstream generation failed: ${String(data.error ?? "unknown")}.`; + if (paymentStatus === "not_charged") { + throw new Error(`${failed} ${note ?? "No payment was taken."}`); + } + // Anything short of an observed refund is bookable: an explicit charged + // status is certain, an absent one is unknown — and unknown books too, + // because the gateway's contract is to say "not_charged" when it refunds. const billing = note ?? - (paymentStatus === "not_charged" - ? "No payment was taken." - : `Billing status: ${paymentStatus ?? "unknown"} — check https://user.blockrun.ai/dashboard/activity${jobId ? ` for job ${jobId}` : ""}.`); - throw new Error(`Upstream generation failed: ${String(data.error ?? "unknown")}. ${billing}`); + `Billing status: ${paymentStatus ?? "unknown"} — check https://user.blockrun.ai/dashboard/activity${jobId ? ` for job ${jobId}` : ""}.`; + throw new BilledJobError(`${failed} ${billing}`, { paidUsd: submitCost, jobId, billing: paymentStatus ? "billed" : "unknown" }); } if (poll.ok && lastStatus === "completed") { // Async media bills at SUBMIT and the polls are free, so the price rides @@ -249,16 +355,19 @@ export async function apiKeyAsyncPost( // poll's header if one ever appears, but fall back to the submit's. return { data, paidUsd: costFrom(poll) ?? submitCost, txHash: receiptFrom(poll), jobId }; } - // 504 is a transient upstream poll timeout on this gateway, same as the - // wallet rails — keep polling rather than abandoning a paid job. - if (!poll.ok && poll.status !== 202 && poll.status !== 504) { - await throwForStatus(poll, `poll ${absolutePollUrl}`); + if (TRANSIENT_POLL_STATUSES.has(poll.status)) { + // Honour Retry-After when the proxy sends one, but never sleep past the + // deadline; the loop's own interval covers the rest. + const wait = Math.min(retryAfterMs(poll), Math.max(0, deadline - Date.now())); + if (wait > 0) await new Promise((r) => setTimeout(r, wait)); + continue; + } + if (!poll.ok && poll.status !== 202) { + throw billed(`${statusErrorMessage(poll, `poll ${absolutePollUrl}`, data)} ${billedNote}`); } } - throw new Error( - `Job did not complete within ${Math.round(pollBudgetMs / 1000)}s (last status: ${lastStatus}). ` + - `It has already been billed to the account${jobId ? `; job id ${jobId}` : ""} — ` + - `check https://user.blockrun.ai/dashboard/activity before submitting again.`, + throw billed( + `Job did not complete within ${Math.round(pollBudgetMs / 1000)}s (last status: ${lastStatus}). ${billedNote}`, ); } diff --git a/src/utils/budget.ts b/src/utils/budget.ts index b24beb1..617d920 100644 --- a/src/utils/budget.ts +++ b/src/utils/budget.ts @@ -205,3 +205,64 @@ export function parseBudgetLimitEnv(raw: string | undefined): number | null { const n = Number(raw.trim().replace(/^\$/, "")); return Number.isFinite(n) && n > 0 ? n : null; } + +// --------------------------------------------------------------------------- +// Quote sanity — pay what you were told, or nothing +// --------------------------------------------------------------------------- +// +// Every manual-402 tool estimates the charge from a published rate table, then +// reads the REAL price off the gateway's 402 before signing. Until now the only +// check on that real price was the budget cap: a quote above the estimate was +// re-reserved and paid. That is right for a token-priced 4K render that the +// table undershoots by a cent, and wrong for what verify:prices found on +// 2026-09-08: the Solana gateway (a separate deployment that can lag Base) +// does not know azure/sora-2 and quotes it as "Seedance 2.0 Pro video +// generation (5s)" at $1.135 — 2.7x the published Sora rate, for a different +// model. The budget cap would have let that through on any wallet with $2. +// +// So: a quote more than QUOTE_TOLERANCE_RATIO above the estimate (and more than +// QUOTE_TOLERANCE_FLOOR_USD above it, so a $0.003 quote against a $0.001 +// estimate is not a "3x") is refused before anything is signed. The estimators +// are verified against live 402s to within $0.001 (`npm run verify:prices`), so +// the honest cases live far inside 1.5x; a legitimate gateway reprice past it +// fails loud until the estimator is updated, which is the safe direction for +// money. Nothing here touches the ledger — a refused quote settles nothing. +export const QUOTE_TOLERANCE_RATIO = 1.5; +export const QUOTE_TOLERANCE_FLOOR_USD = 0.02; + +export class QuoteMismatchError extends Error { + readonly quotedUsd: number; + readonly estimateUsd: number; + constructor(message: string, quotedUsd: number, estimateUsd: number) { + super(message); + this.name = "QuoteMismatchError"; + this.quotedUsd = quotedUsd; + this.estimateUsd = estimateUsd; + } +} + +/** + * Throws QuoteMismatchError when the gateway's authoritative quote is far above + * what the caller told the user to expect. `null` quotes are not judged here — + * callers already fail closed on an unreadable amount. The message ends with + * "no charge was made" so formatError() does not append funding advice. + */ +export function assertQuoteNearEstimate( + quotedUsd: number | null | undefined, + estimateUsd: number, + opts: { what: string; quotedFor?: string; hint?: string }, +): void { + if (typeof quotedUsd !== "number" || !Number.isFinite(quotedUsd)) return; + if (!(estimateUsd > 0)) return; // a $0 estimate means "free": nothing to compare + const ratio = quotedUsd / estimateUsd; + if (ratio <= QUOTE_TOLERANCE_RATIO || quotedUsd - estimateUsd <= QUOTE_TOLERANCE_FLOOR_USD) return; + const labelled = opts.quotedFor ? ` — the gateway labels that quote "${opts.quotedFor}"` : ""; + throw new QuoteMismatchError( + `The gateway quoted $${quotedUsd.toFixed(4)} for ${opts.what}, but this tool expected about $${estimateUsd.toFixed(4)} ` + + `(${ratio.toFixed(1)}x the published rate)${labelled}. Refusing to sign it — no charge was made. ` + + `A gap this large means the gateway repriced the model or substituted a different one.` + + (opts.hint ? ` ${opts.hint}` : ""), + quotedUsd, + estimateUsd, + ); +} diff --git a/src/utils/constants.ts b/src/utils/constants.ts index 8271292..f4191c6 100644 --- a/src/utils/constants.ts +++ b/src/utils/constants.ts @@ -38,7 +38,9 @@ export const BASE_RPC_URLS = [ // presence in the catalogue is NECESSARY BUT NOT SUFFICIENT for health; // absence from it is NOT SUFFICIENT for death — probe before deleting. // -// OpenAI (27): gpt-5.6-sol ($5/$30, 1M, deepest reasoning), gpt-5.6-terra +// OpenAI (28): gpt-6-astra ($10/$50, 1M — the GPT-6 flagship; listed by +// 2026-09-08 ABOVE the $5/$30 default, so it carries a CHAT_PRICE_PER_MTOKEN +// row), gpt-5.6-sol ($5/$30, 1M, deepest reasoning), gpt-5.6-terra // ($2/$12, 1M — the balanced default; CUT from $2.5/$15), gpt-5.6-luna // ($0.2/$1.2, 1M, no reasoning; CUT from $1/$6), plus the 2026-08 "pro // reasoning mode" trio: gpt-5.6-sol-pro ($5/$30), gpt-5.6-terra-pro ($1/$6 — @@ -50,10 +52,13 @@ export const BASE_RPC_URLS = [ // gpt-5.3-codex ($1.75/$14), gpt-5.2-pro ($21/$168), gpt-5.4-mini, // gpt-5-mini, gpt-5.4-nano, gpt-4.1{,-mini,-nano}, gpt-4o{,-mini}, // o1 ($15/$60), o3 ($2/$8), o3-mini, o4-mini -// Anthropic (9): claude-opus-5 ($5/$25, 1M, 128k out — newest Opus, +// Anthropic (10): claude-fable-5.1 ($10/$50, 1M — listed by 2026-09-08 ABOVE +// the default, so it carries a price row), claude-opus-5 ($5/$25, 1M, 128k out — newest Opus, // step-change over 4.8 at the same price; live-probed 2026-08-12), // claude-opus-4.8 ($5/$25, 1M), claude-fable-5 ($10/$50, 1M), -// claude-opus-4.7 ($5/$25, 1M), claude-sonnet-5 ($3/$15, 1M), +// claude-opus-4.7 ($5/$25, 1M), claude-sonnet-5 ($2/$10, 1M — CUT from +// $3/$15; both gateways 2026-09-08. On the native path the price row IS the +// ledger, so the stale row over-booked every sonnet-5 call 1.5x), // claude-opus-4.5, claude-sonnet-4.6, claude-sonnet-4.5, claude-haiku-4.5 // Google (9): gemini-3.1-pro ($2/$12), gemini-3.6-flash ($1.5/$7.5, thinking — // newest Flash), gemini-3.5-flash ($1.5/$9 — REPRICED from the $0.5/$3 this @@ -94,6 +99,18 @@ export const BASE_RPC_URLS = [ // aliases on Base (it served itself in 0.97s; July saw it alias to // gpt-oss-120b there) — so the reason it was documented-but-unrouted is // gone, and it joins free[]. +// 2026-09-08 CATALOGUE (listing only — no POST probe was run): the +// billing_mode:"free" set on both chains is nemotron-3-nano-omni, +// llama-3.2-11b-vision, nemotron-3-ultra-550b, nemotron-3.5-lightning +// (available:false on Base that day) — plus, for the first time, two $0 +// models OUTSIDE nvidia/: cohere/north-mini-code and poolside/laguna-xs-2.1. +// That is why FREE_CHAT_MODELS exists below: "free" was a vendor-prefix test, +// and it refused those two at an exhausted budget. Solana additionally lists +// muse-glimmer-30b and gemma-4-31b. step-3.7-flash, mistral-nemotron, +// gpt-oss-20b and both nemotron-nano-* are no longer listed anywhere — and by +// the rule above that is NOT a death certificate (gpt-oss-120b has been +// hidden-alive since July). They stay routed at the tail of free[] until a +// realistic-prompt POST probe reads their response `model` field. export const MODEL_TIERS = { fast: ["google/gemini-3.5-flash", "google/gemini-2.5-flash", "openai/gpt-5.6-luna", "google/gemini-3.5-flash-lite", "openai/gpt-5-mini", "deepseek/deepseek-chat", "google/gemini-3-flash-preview"], balanced: ["openai/gpt-5.6-terra", "anthropic/claude-sonnet-5", "moonshot/kimi-k3", "google/gemini-3.1-pro", "xai/grok-4.5", "openai/gpt-5.5"], @@ -130,17 +147,68 @@ export const MODEL_TIERS = { // healthy; the same model on a realistic 1.5K-token prompt took 123.2s. That // is the trap the gateway's own probe script added a --real mode for. Never // health-check a free model with a 16-token ping. - free: ["nvidia/gpt-oss-120b", "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "nvidia/step-3.7-flash", "nvidia/mistral-nemotron", "nvidia/gpt-oss-20b", "nvidia/nemotron-nano-12b-v2-vl", "nvidia/nemotron-nano-9b-v2"], + // + // 2026-09-08 order, three bands, from the live catalogue (listing evidence + // only — see the NVIDIA note above for what was and was not probed): + // 1. gpt-oss-120b — hidden-alive, the gateway's own free fallback; and + // nemotron-3-nano-omni — listed, available, served itself on 2026-08-12. + // 2. Listed billing_mode:"free" on BOTH chains and available: the two new + // NVIDIA entries and the first two non-NVIDIA free models. Unprobed for + // latency, so they sit behind the proven pair, not ahead of it. + // 3. The four delisted entries. Delisting tells you nothing either way; + // each is bounded by FREE_MODEL_TIMEOUT_MS and the loop by + // FREE_TIER_DEADLINE_MS, so a dead tail costs time, never money. + // Remove them only on a POST probe that shows aliasing or a crawl. + // Skipped on purpose: nemotron-3.5-lightning (available:false on Base), + // muse-glimmer-30b and gemma-4-31b (Solana catalogue only) — routing has to + // hold on both chains. They are still in FREE_CHAT_MODELS, so an explicit + // call to one reserves $0 like any other free id. + free: [ + "nvidia/gpt-oss-120b", "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", + "nvidia/llama-3.2-11b-vision", "nvidia/nemotron-3-ultra-550b", "cohere/north-mini-code", "poolside/laguna-xs-2.1", + "nvidia/step-3.7-flash", "nvidia/mistral-nemotron", "nvidia/gpt-oss-20b", "nvidia/nemotron-nano-12b-v2-vl", "nvidia/nemotron-nano-9b-v2", + ], coding: ["anthropic/claude-opus-5", "openai/gpt-5.3-codex", "moonshot/kimi-k3", "xai/grok-build-0.1", "zai/glm-5.2", "qwen/qwen3.7-max", "anthropic/claude-sonnet-5"], glm: ["zai/glm-5", "zai/glm-5.2", "zai/glm-5.1", "zai/glm-5-turbo"], } as const; export type RoutingMode = keyof typeof MODEL_TIERS; +/** + * Every chat model the gateway bills at $0 — the set the budget gate consults to + * reserve nothing, and the truth the routing tier's free[] has to be a subset of. + * + * This used to be a vendor test, `startsWith("nvidia/")`. It was true of the + * whole free tier on 2026-08-12 and is wrong since the 2026-09-08 catalogue, + * which bills cohere/north-mini-code and poolside/laguna-xs-2.1 at $0 on both + * gateways: an explicit call to either reserved the $5/$30 DEFAULT, so an + * exhausted budget refused a free call and confirm-spend asked a human to + * approve a phantom charge — the false refusal the bare-`gpt-oss-120b` + * canonicalisation fixed, reopened on the vendor axis. Membership is by id. + * + * The dangerous direction is the other one: a member that STARTS costing money + * gets a $0 reserve for a paid call, which is the total gate bypass this file + * spends so many words preventing. So `npm run verify:prices` checks every + * member against the live catalogue and fails if one is priced. A Set needs no + * hasOwn guard — there are no prototype keys to leak through `.has`. + * + * Not every member is routed: MODEL_TIERS.free wants a both-chain, latency- + * ordered list (see its notes); this set only says what is free. + */ +export const FREE_CHAT_MODELS: ReadonlySet = new Set([ + ...MODEL_TIERS.free, + // Live billing_mode:"free" on 2026-09-08 but deliberately not routed. + "nvidia/nemotron-3.5-lightning", // available:false on Base that day + "nvidia/muse-glimmer-30b", // Solana catalogue only + "nvidia/gemma-4-31b", // Solana catalogue only +]); + /** * $/M input and output for every model a routing tier can resolve to, plus every * catalog model priced ABOVE the DEFAULT_CHAT_PRICE an explicit `model` falls - * back to. Read off the live GET /v1/models on 2026-08-13. + * back to. Read off the live GET /v1/models on 2026-08-13; gpt-6-astra and + * claude-fable-5.1 added from the 2026-09-08 catalogue, after both had sat + * above the default with no row for weeks (see the block comment below). * * This exists because the budget gate used to reserve chat against two hardcoded * constants — "$5/M input" and "4 chars per token" — and BOTH were wrong at the @@ -165,17 +233,27 @@ export type RoutingMode = keyof typeof MODEL_TIERS; * * Keep this in step with MODEL_TIERS: a tier member with no entry here reserves * DEFAULT_CHAT_PRICE, which is correct for everything at or below $5/$30 and - * SHORT for the five models above it. `npm run verify:prices` probes one row per - * tier against the live 402 so drift shows up as a failure, not as a surprise - * invoice. + * SHORT for the seven models above it. `npm run verify:prices` probes one row + * per tier against the live 402, and sweeps GET /v1/models on both gateways for + * any available model priced above the default with no row here — so drift in + * EITHER direction (a reprice, or a new model landing above the line) shows up + * as a failure, not as a surprise invoice. */ export const CHAT_PRICE_PER_MTOKEN: Record = { - // Above the default — the five that made the gate unsafe. Reachable as an - // explicit `model` as well as through powerful/reasoning. + // Above the default — the ids that make the gate unsafe without a row. + // Reachable as an explicit `model` as well as through powerful/reasoning. + // Seven as of 2026-09-08: the five pro-tier outliers the table was built for, + // plus the two flagships that landed above $5/$30 afterwards and went + // unnoticed for weeks — no row, no test, and no sweep, so an explicit + // model:"openai/gpt-6-astra" reserved at half its real rate and the + // confirm-spend prompt showed a human the same wrong number. The catalogue + // sweep in scripts/verify-prices.ts now fails on the eighth. "openai/gpt-5.4-pro": { input: 30, output: 180 }, "openai/gpt-5.5-pro": { input: 30, output: 180 }, "openai/gpt-5.2-pro": { input: 21, output: 168 }, "openai/o1": { input: 15, output: 60 }, + "openai/gpt-6-astra": { input: 10, output: 50 }, + "anthropic/claude-fable-5.1": { input: 10, output: 50 }, "anthropic/claude-fable-5": { input: 10, output: 50 }, // At or below the default — listed so the CHEAP tiers reserve their own real // rate instead of the $5/$30 worst case, which would price a qwen3.7-flash @@ -184,7 +262,12 @@ export const CHAT_PRICE_PER_MTOKEN: Record = Object.fromEntries( - [...Object.keys(CHAT_PRICE_PER_MTOKEN), ...Object.values(MODEL_TIERS).flat()] + [...Object.keys(CHAT_PRICE_PER_MTOKEN), ...Object.values(MODEL_TIERS).flat(), ...FREE_CHAT_MODELS] .filter((id: string) => id.includes("/")) .map((id: string) => [id.slice(id.indexOf("/") + 1), id]), ); @@ -302,7 +393,7 @@ export const TIER_WORST_PRICE: Record Object.hasOwn(CHAT_PRICE_PER_MTOKEN, id) ? CHAT_PRICE_PER_MTOKEN[id] - : (id.startsWith("nvidia/") ? { input: 0, output: 0 } : DEFAULT_CHAT_PRICE), + : (FREE_CHAT_MODELS.has(id) ? { input: 0, output: 0 } : DEFAULT_CHAT_PRICE), ); return [mode, { input: Math.max(...rates.map((r) => r.input)), diff --git a/src/utils/errors.ts b/src/utils/errors.ts index ea2634c..9c6b2b1 100644 --- a/src/utils/errors.ts +++ b/src/utils/errors.ts @@ -19,6 +19,14 @@ export function extractErrorMessage(err: unknown): string { // Common gateway error shape: { error, message, hint, missing_params? } const parts: string[] = []; if (typeof b.message === "string") parts.push(b.message); + // @blockrun/llm >= 3.15.1 (blockrun-llm-ts#39) keeps the gateway's own + // `message` — the field that names the cause AND says whether money moved, + // e.g. "Predexon 500: … (payment NOT charged)" — under `detail`, because + // the sanitizer already uses `message` for the top-level `error` string. + // Without this line that text is dropped a second time here, and + // formatError() below has no evidence to say nothing was charged — all it + // can echo is the SDK's "after payment" prefix (blockrun-mcp#132). + if (typeof b.detail === "string" && b.detail !== b.message) parts.push(b.detail); if (typeof b.hint === "string") parts.push(`Hint: ${b.hint}`); if (Array.isArray(b.missing_params) && b.missing_params.length) { parts.push(`Missing: ${b.missing_params.join(", ")}`); @@ -47,6 +55,24 @@ export function isPaymentRejectionError(message: string): boolean { return m.includes("insufficient") || m.includes("balance") || m.includes("rejected"); } +/** + * True when `message` carries a 5xx that READS as an HTTP status. A bare + * three-digit match is far too loose: LLM errors are full of incidental + * 5xx-shaped numbers ("max_tokens 512 is above the limit", "embedding dimension + * 512"), and telling the user to wait out a temporary outage hides a real + * validation bug. Either the number is directly labelled as a status ("error + * 500", "status code 503", "http 502") — adjacency matters, so "context length + * 512 exceeded" does not qualify — or it carries a standard HTTP reason phrase. + * Shared with the route-specific formatters so they cannot drift looser. + */ +export function hasLabelledServerStatus(message: string): boolean { + const m = message.toLowerCase(); + // "payment" is a label too: the SDK's post-402 prefix is "API error after + // payment: 502", where the word before the number is "payment", not "error". + return /(?:status(?:\s*code)?|http|error|payment)\s*[:=]?\s*5[0-9]{2}(?:$|[^0-9.])/.test(m) || + /(?:^|[^0-9.])5[0-9]{2}:?\s+(?:internal|server error|bad gateway|service unavailable|gateway time)/.test(m); +} + /** * Format an error for return to the caller, appending actionable guidance for * the three common failure classes (upstream model unavailable, server blip, @@ -89,22 +115,28 @@ export function formatError(message: string, opts?: { altModels?: string }): str // actionable client errors, not transient server outages, so they no longer // get retry guidance. // - // A 5xx must LOOK like an HTTP status to count. A bare three-digit match is - // far too loose here: LLM errors are full of incidental 5xx-shaped numbers - // ("max_tokens 512 is above the limit", "embedding dimension 512"), and - // telling the user to wait out a temporary outage hides a real validation bug. - // Either the number is directly labelled as a status ("error 500", - // "status code 503", "http 502") — adjacency matters, so "context length 512 - // exceeded" does not qualify — or it carries a standard HTTP reason phrase. - const has5xxStatus = - /(?:status(?:\s*code)?|http|error)\s*[:=]?\s*5[0-9]{2}(?:$|[^0-9.])/.test(msgLower) || - /(?:^|[^0-9.])5[0-9]{2}:?\s+(?:internal|server error|bad gateway|service unavailable|gateway time)/.test(msgLower); + // See hasLabelledServerStatus: a 5xx must LOOK like an HTTP status to count. + const has5xxStatus = hasLabelledServerStatus(msgLower); // A post-payment failure with no parseable status is still an upstream // failure, not an empty wallet — without this it falls through to the // "payment" keyword branch and wrongly tells the user to fund. const isServerError = has5xxStatus || (msgLower.includes("api error after payment") && !isPostPaymentClientError); + // 501 is the gateway saying "we do not serve this" — the equity price/history + // routes have answered it before any payment since 2026-09-05 (licensing; + // blockrun#517). It is in the 5xx range but it is not an outage, so "try again + // in a few minutes" is wrong advice and hides that the product line is gone. + // Same labelling rule as has5xxStatus: the number must read as a status, so + // "batch of 501 items" does not qualify. The "nothing was charged" claim is + // only safe when the 501 arrived BEFORE payment: a post-payment 501 means the + // gateway settled and then upstream refused, and this formatter has no + // endpoint context to know whether the nonce was released. + const isNotServed = + /(?:status(?:\s*code)?|http|error|payment)\s*[:=]?\s*501(?:$|[^0-9.])/.test(msgLower) || + /(?:^|[^0-9.])501:?\s+not implemented/.test(msgLower); + const isNotServedPrePayment = isNotServed && !msgLower.includes("api error after payment"); + const altHint = opts?.altModels ? ` (e.g. ${opts.altModels})` : ""; let errorText = `Error: ${message}`; @@ -113,10 +145,27 @@ export function formatError(message: string, opts?: { altModels?: string }): str (opts?.altModels ? `. Try a different model${altHint} — it should work right away.` : `. Try a different model, or retry shortly.`); + } else if (isNotServed) { + errorText += `\n\nThe gateway does not serve this endpoint (501 Not Implemented). This is not a` + + `\ntransient outage — retrying will not help` + + (isNotServedPrePayment + ? `, and nothing was charged.` + : `. Check blockrun_wallet action:"report" to see whether this call settled.`); } else if (isServerError) { errorText += `\n\nThis is a temporary API issue. The API may be experiencing problems.` + `\nTry again in a few minutes` + (opts?.altModels ? `, or use a different model${altHint}.` : `.`); + // The gateway's own words, restated as guidance. The SDK labels every + // post-402 failure "API error after payment", and until now the only thing + // `explicitlyUncharged` did was suppress the funding footer — which this + // branch, tested first, already made unreachable for a 5xx. So a + // "(payment NOT charged)" 5xx read as "after payment … try again", with + // nothing in the tool's voice saying whether money moved (blockrun-mcp#132). + // Only the gateway's marker earns this line; the formatter never invents a + // settlement claim of its own. + if (explicitlyUncharged) { + errorText += `\nThe gateway reported that this call was not settled — nothing was charged.`; + } } else if (isPaymentError) { const chain = getChain(); const network = chain === "solana" ? "Solana" : "Base"; diff --git a/src/utils/key-leak-scanner.ts b/src/utils/key-leak-scanner.ts index feab9f4..8bf7afe 100644 --- a/src/utils/key-leak-scanner.ts +++ b/src/utils/key-leak-scanner.ts @@ -8,6 +8,14 @@ * which put the private key in ~/.claude.json (plaintext, 0644, often * synced to iCloud/Dropbox/Time Machine). * + * One location is NOT a leak: `mcpServers..env.BLOCKRUN_WALLET_KEY` / + * `SOLANA_WALLET_KEY`. That is the documented env override (README env table, + * server.template.json `environmentVariables`), and on Claude Code the only + * way to set it is `claude mcp add -e BLOCKRUN_WALLET_KEY=0x… -s user`, which + * writes exactly that path into ~/.claude.json. It still lives in a synced + * plaintext file, so it earns a short note pointing at the safer stores — but + * not the "treat as compromised, rotate" banner the hosted-auth paste gets. + * * See: https://github.com/BlockRunAI/blockrun-mcp-server/issues/1 */ @@ -51,99 +59,197 @@ function looksLikeSolanaSecretKeyArray(value: unknown): boolean { ); } -interface Finding { +/** + * `leak` — a key somewhere it was never meant to be (the hosted-auth header + * paste, a stray raw key): rotate. + * `env-override` — the documented `mcpServers.*.env.{BLOCKRUN,SOLANA}_WALLET_KEY` + * override: works, but a synced plaintext file is a weaker store than + * ~/.blockrun/.session or the OS keychain. Note, do not alarm. + */ +export type FindingKind = "leak" | "env-override"; + +export interface Finding { file: string; path: string; // JSON path like "mcpServers.blockrun.headers.X-Wallet-Key" + kind: FindingKind; } -function walk( - obj: unknown, - file: string, - jsonPath: string, - out: Finding[], -): void { +/** The env var names the README, server.template.json and the setup skill document. */ +const DOCUMENTED_KEY_ENV_VARS = new Set(["BLOCKRUN_WALLET_KEY", "SOLANA_WALLET_KEY"]); + +/** + * True when `segments` ends in `mcpServers..env.`. + * Segment-based rather than a regex over the dotted path so a server name that + * itself contains a dot cannot slip the check. Matches user scope + * (`mcpServers.…`) and Claude Code's project scope (`projects..mcpServers.…`) + * alike, and every JSON client (Claude Desktop, Cursor, Windsurf) uses the same + * `mcpServers..env` shape. + */ +function isDocumentedEnvOverride(segments: string[]): boolean { + const n = segments.length; + return ( + n >= 4 && + segments[n - 4] === "mcpServers" && + segments[n - 2] === "env" && + DOCUMENTED_KEY_ENV_VARS.has(segments[n - 1]) + ); +} + +function displayPath(segments: string[]): string { + return segments.reduce((acc, s) => (/^\[\d+\]$/.test(s) ? acc + s : acc ? `${acc}.${s}` : s), ""); +} + +function walk(obj: unknown, file: string, segments: string[], out: Finding[]): void { if (obj === null || typeof obj !== "object") return; if (Array.isArray(obj)) { if (looksLikeSolanaSecretKeyArray(obj)) { - out.push({ file, path: jsonPath || "(root)" }); + out.push({ file, path: displayPath(segments) || "(root)", kind: "leak" }); } - obj.forEach((v, i) => walk(v, file, `${jsonPath}[${i}]`, out)); + obj.forEach((v, i) => { + const next = [...segments, `[${i}]`]; + // A string element is a leaf — walk() returns at once for primitives — + // so it must be checked HERE. Without this a key passed as an `args` + // element (`"args": ["-e", "0x…"]`) was never seen. + if (looksLikeRawPrivateKey(v)) out.push({ file, path: displayPath(next), kind: "leak" }); + walk(v, file, next, out); + }); return; } for (const [k, v] of Object.entries(obj as Record)) { - const next = jsonPath ? `${jsonPath}.${k}` : k; + const next = [...segments, k]; // Heuristic: look at header-ish fields first. A key/secret-named field uses // the permissive matcher so a bare 64-hex key is caught too. - if (/wallet[-_ ]?key|private[-_ ]?key|secret/i.test(k) && looksLikeNamedSecretValue(v)) { - out.push({ file, path: next }); - } else if (looksLikeRawPrivateKey(v)) { - // Also catch untagged values that happen to be raw keys - out.push({ file, path: next }); + const named = /wallet[-_ ]?key|private[-_ ]?key|secret/i.test(k) && looksLikeNamedSecretValue(v); + // Also catch untagged values that happen to be raw keys. + if (named || looksLikeRawPrivateKey(v)) { + out.push({ file, path: displayPath(next), kind: isDocumentedEnvOverride(next) ? "env-override" : "leak" }); } walk(v, file, next, out); } } +/** Scan one parsed config object. Exported for tests; `warnOnLeakedKeys` is the caller. */ +export function findKeyLeaks(data: unknown, file: string): Finding[] { + const out: Finding[] = []; + walk(data, file, [], out); + return out; +} + function scanFile(file: string): Finding[] { try { if (!fs.existsSync(file)) return []; const raw = fs.readFileSync(file, "utf-8"); - const data = JSON.parse(raw) as unknown; - const out: Finding[] = []; - walk(data, file, "", out); - return out; + return findKeyLeaks(JSON.parse(raw) as unknown, file); } catch { return []; } } /** - * Scan well-known config files for leaked wallet keys. Returns true if any - * findings were printed; caller may choose to exit if strict mode is desired. + * The MCP config files this package documents an install path for (README + * "Install" table, skills/blockrun-setup): Claude Code's ~/.claude.json, Claude + * Desktop, Cursor and Windsurf. Codex (~/.codex/config.toml) is TOML and is not + * scanned. Windows paths come from %APPDATA% when set — it is the variable the + * docs name and a redirected profile does not have to sit under the home dir — + * with the conventional `AppData/Roaming` as the fallback. Deduplicated because + * the fallback and %APPDATA% usually coincide. */ -export function warnOnLeakedKeys(): boolean { - const home = os.homedir(); +export function configFileCandidates(home: string = os.homedir(), env: NodeJS.ProcessEnv = process.env): string[] { + const appData = env.APPDATA && env.APPDATA.trim() ? env.APPDATA : path.join(home, "AppData", "Roaming"); const candidates = [ + // Claude Code (user scope; project-scoped servers live in the same file) path.join(home, ".claude.json"), + // Claude Desktop — macOS, Linux (Electron userData is ~/.config/, + // and the product name is capitalised; the lowercase spelling is kept for + // anyone who followed an older guide), Windows path.join(home, "Library", "Application Support", "Claude", "claude_desktop_config.json"), + path.join(home, ".config", "Claude", "claude_desktop_config.json"), path.join(home, ".config", "claude", "claude_desktop_config.json"), - path.join(home, "AppData", "Roaming", "Claude", "claude_desktop_config.json"), + path.join(appData, "Claude", "claude_desktop_config.json"), + // Cursor + path.join(home, ".cursor", "mcp.json"), + path.join(appData, "Cursor", "mcp.json"), + // Windsurf + path.join(home, ".codeium", "windsurf", "mcp_config.json"), + path.join(home, ".config", ".codeium", "windsurf", "mcp_config.json"), + path.join(appData, "Codeium", "windsurf", "mcp_config.json"), ]; + return [...new Set(candidates)]; +} + +export interface WarnOptions { + /** Files to scan; defaults to `configFileCandidates()`. */ + files?: string[]; + /** Line sink; defaults to console.error (stderr is the MCP log channel). */ + log?: (line: string) => void; +} + +/** + * Scan well-known config files for wallet keys. Prints the rotate-your-wallet + * banner for a real leak and a short store-it-somewhere-safer note for the + * documented env override. Returns true only when a real leak was printed; + * the caller may choose to exit on that if strict mode is desired. + */ +export function warnOnLeakedKeys(opts: WarnOptions = {}): boolean { + const files = opts.files ?? configFileCandidates(); + const log = opts.log ?? ((line: string) => console.error(line)); const findings: Finding[] = []; - for (const f of candidates) findings.push(...scanFile(f)); + for (const f of files) findings.push(...scanFile(f)); + + const leaks = findings.filter((f) => f.kind === "leak"); + const overrides = findings.filter((f) => f.kind === "env-override"); - if (findings.length === 0) return false; + if (leaks.length > 0) printLeakBanner(leaks, log); + if (overrides.length > 0) printEnvOverrideNote(overrides, log); + return leaks.length > 0; +} + +function printLeakBanner(findings: Finding[], log: (line: string) => void): void { const bar = "═".repeat(72); - console.error(""); - console.error(`\x1b[31m${bar}`); - console.error(" 🚨 WALLET PRIVATE KEY DETECTED IN CONFIG FILE"); - console.error(bar + "\x1b[0m"); - console.error(""); - console.error(" Your config contains what looks like a raw wallet private key."); - console.error(" Private keys should NEVER be stored in these files — they get"); - console.error(" backed up to iCloud / Dropbox / Time Machine, synced across"); - console.error(" machines, and readable by anything that can read your config."); - console.error(""); - console.error(" Found in:"); + log(""); + log(`\x1b[31m${bar}`); + log(" 🚨 WALLET PRIVATE KEY DETECTED IN CONFIG FILE"); + log(bar + "\x1b[0m"); + log(""); + log(" Your config contains what looks like a raw wallet private key."); + log(" Private keys should NEVER be stored in these files — they get"); + log(" backed up to iCloud / Dropbox / Time Machine, synced across"); + log(" machines, and readable by anything that can read your config."); + log(""); + log(" Found in:"); for (const f of findings) { - console.error(` · ${f.file}`); - console.error(` at: ${f.path}`); + log(` · ${f.file}`); + log(` at: ${f.path}`); } - console.error(""); - console.error(" RECOMMENDED ACTIONS:"); - console.error(" 1. Treat this key as compromised. Rotate your wallet:"); - console.error(" - Create a new wallet"); - console.error(" - Transfer remaining USDC to the new address"); - console.error(" - Retire the old key"); - console.error(" 2. Remove the X-Wallet-Key entries from your config."); - console.error(" 3. Reconnect using the local package (signs locally, key"); - console.error(" never leaves your machine):"); - console.error(" claude mcp remove blockrun"); - console.error(" claude mcp add blockrun -s user -- npx -y @blockrun/mcp@latest"); - console.error(""); - console.error(" Details: https://github.com/BlockRunAI/blockrun-mcp-server/issues/1"); - console.error(""); - return true; + log(""); + log(" RECOMMENDED ACTIONS:"); + log(" 1. Treat this key as compromised. Rotate your wallet:"); + log(" - Create a new wallet"); + log(" - Transfer remaining USDC to the new address"); + log(" - Retire the old key"); + log(" 2. Remove the X-Wallet-Key entries from your config."); + log(" 3. Reconnect using the local package (signs locally, key"); + log(" never leaves your machine):"); + log(" claude mcp remove blockrun"); + log(" claude mcp add blockrun -s user -- npx -y @blockrun/mcp@latest"); + log(""); + log(" Details: https://github.com/BlockRunAI/blockrun-mcp-server/issues/1"); + log(""); +} + +/** + * The documented override, working as documented. Said once per start, in + * four lines, with no rotate advice: the key was placed there on purpose and + * is not known to anyone else. What it does deserve is the reminder that the + * file is plaintext and synced, and where the safer stores are. + */ +function printEnvOverrideNote(findings: Finding[], log: (line: string) => void): void { + log("[BlockRun] Your wallet key is set as an env var in an MCP client config file:"); + for (const f of findings) log(`[BlockRun] · ${f.file} at ${f.path}`); + log( + "[BlockRun] That works, but the file is plaintext and usually synced (iCloud / Dropbox / Time Machine). " + + "Prefer the default ~/.blockrun/.session (0600) or the OS keychain (BLOCKRUN_KEYCHAIN=auto), then drop the env entry.", + ); } diff --git a/src/utils/keychain.ts b/src/utils/keychain.ts index f80383f..a04638c 100644 --- a/src/utils/keychain.ts +++ b/src/utils/keychain.ts @@ -3,12 +3,19 @@ // OS keychain storage for wallet private keys. // // Background: the wallet key lives at ~/.blockrun/.session as plaintext (mode -// 0600). File permissions stop other UNIX users, but not anything running as -// you — a malicious postinstall script, a backup agent that syncs the home -// directory to iCloud/Dropbox, or a leaky log collector all read it trivially. -// The same class of leak already bit us once through ~/.claude.json, which is -// why utils/key-leak-scanner.ts exists. The keychain moves the secret behind -// an OS-mediated API instead of a readable path. +// 0600). File permissions stop other UNIX users, but not anything that reads +// the home directory as data — a backup agent syncing to iCloud/Dropbox, a +// disk image, a dotfile or log slurper. The same class of leak already bit us +// once through ~/.claude.json, which is why utils/key-leak-scanner.ts exists. +// The keychain moves the secret behind an OS-mediated API instead of a +// readable path, so it protects the key AT REST. +// +// What it does NOT do: defend against code running as you. `security +// add-generic-password` without -T/-A grants the creating application (here, +// /usr/bin/security itself) access, and any process running as the user can +// run `security find-generic-password -w` / `secret-tool lookup` and read the +// value back — a malicious postinstall script included. Do not describe the +// keychain as protection against same-user code; it is not. // // Technique credit: the `security -i` approach below (and the 128-byte // truncation gotcha it avoids) is adapted from Circle's CLI, Apache-2.0. diff --git a/src/utils/markets-validation.ts b/src/utils/markets-validation.ts index 3aff04d..d01ce1c 100644 --- a/src/utils/markets-validation.ts +++ b/src/utils/markets-validation.ts @@ -1,3 +1,4 @@ +import { hasLabelledServerStatus } from "./errors.js"; import { normalizeClassifyPath } from "./path-safety.js"; /** @@ -152,3 +153,76 @@ export function validateMarketRequest( return null; } + +// --------------------------------------------------------------------------- +// Degraded upstream routes — known to fail, NOT charged +// --------------------------------------------------------------------------- +// +// Every sports/* path has returned a consistent Predexon 500 ("An unexpected +// error occurred") since 2026-08-04 — re-verified live 2026-09-08 (#132). The +// gateway marks them `status: "degraded"` in its predexon.ts: still routed for +// anyone who knows the path, withheld from openapi.json and the x402 manifest, +// and on an upstream 5xx it releases the payment nonce, so nothing settles. +// +// Not a pre-payment block like markets/listings above: that one is a 410 sunset +// and settles before failing, this one is an upstream bug that may recover, and +// the gateway is the authority on whether it has. What we own is the wording. +// The SDK reduced the gateway's "(payment NOT charged)" body to +// `API error after payment: 502`, which asserts a charge that did not happen. +export const DEGRADED_SPORTS_SINCE = "2026-08-04"; + +// The one remedy that 402s today on both gateways. Verified 2026-09-08 with +// unauthenticated GETs: markets/search?q=, polymarket/events and kalshi/markets +// all quote a 402; bare `markets`, `outcomes/:id` and `matching-markets` were +// removed upstream 2026-08-04 and 404 BEFORE payment ("Unknown Predexon +// endpoint"), and no live /v1/pm route accepts a `league` param — which is what +// 0.48.1 (never released — folded into 0.49.0) shipped as the steer, so it failed on first use. +const SPORTS_REMEDY = + `For sports odds use "markets/search" with params { q: "NBA" } (every venue in one call), ` + + `"polymarket/events" with params { search: "NBA" }, or "kalshi/markets".`; + +/** + * The gateway's own words for "the payment nonce was released". Only its + * `!upstreamResponse.ok` branch releases, and only that branch writes + * "(payment NOT charged)" / "Upstream provider error" into the body; the + * catch-all 500 ("Internal server error") deliberately does not release + * because settle ran in the same try, and a Vercel 504 after settle is + * text/plain. A 5xx status alone therefore proves nothing about money — + * the phrase does. Mirrors formatError's `explicitlyUncharged` (errors.ts), + * which is not exported; kept local so this file owns its own wording. + */ +function carriesUnchargedEvidence(message: string): boolean { + return /not charged|no charge was made|no payment was made|upstream provider error/i.test(message); +} + +export function isDegradedSportsPath(path: string): boolean { + // Same normalizer as every other rule here (query strip → decode → tab strip), + // so "sports%2Fcategories" gets the same wording as the plain path. + const clean = normalizeMarketPath(path); + return clean === "sports" || clean.startsWith("sports/"); +} + +/** + * Returns the full user-facing error text for a sports/* upstream failure, or + * null when the failure is not the known outage (a 4xx, or a non-sports path) + * so the caller falls back to the generic formatter. + * + * "Nothing was charged" is asserted only when the message carries the + * gateway's own release evidence. Any other labelled 5xx on a sports path is + * still the outage — same explanation, same steer — but money is hedged the + * way formatError hedges a post-payment 501: point at the ledger. + */ +export function describeDegradedSportsFailure(path: string, message: string): string | null { + if (!isDegradedSportsPath(path)) return null; + // Same labelled-status rule as formatError, so "501 items" in an upstream + // 4xx body cannot be mistaken for the outage and sold as "not charged". + if (!hasLabelledServerStatus(message)) return null; + const money = carriesUnchargedEvidence(message) + ? `it released the payment when upstream failed — nothing was charged for this call.` + : `it releases the payment when upstream fails, but this response does not carry the gateway's ` + + `"payment NOT charged" confirmation. Check blockrun_wallet action:"report" to see whether this call settled.`; + return `Error: ${message}\n\n` + + `Predexon's sports/* routes have returned an upstream 500 on every call since ${DEGRADED_SPORTS_SINCE}. ` + + `The gateway still routes them but no longer advertises them, and ${money}\n` + + `Retrying will not help until Predexon repairs the route. ${SPORTS_REMEDY}`; +} diff --git a/src/utils/path-safety.ts b/src/utils/path-safety.ts index 58ea27a..d676ee1 100644 --- a/src/utils/path-safety.ts +++ b/src/utils/path-safety.ts @@ -1,11 +1,11 @@ // src/utils/path-safety.ts // -// Guards for the path-based passthrough tools (rpc, surf, modal, phone, exa, +// Guards for the path-based passthrough tools (rpc, modal, phone, exa, // search, defi, markets). They build a gateway endpoint by concatenating a // caller-supplied slug/path onto a fixed namespace prefix, then hand the string // to fetch(). The WHATWG URL parser collapses dot-segments BEFORE the request is // sent, so a `..` segment escapes the namespace — e.g. -// `/v1/surf/` + `../../v1/modal/sandbox/create` -> `/v1/modal/sandbox/create` +// `/v1/exa/` + `../../v1/modal/sandbox/create` -> `/v1/modal/sandbox/create` // which defeats the per-tool budget pre-check and profile scoping. These helpers // reject the traversal shapes while still allowing unknown-but-wellformed slugs. @@ -26,10 +26,11 @@ * parsing — so `..` is not a `..` segment to a naive equality check, but IS * one by the time fetch() resolves it. That gap was exploitable: * - * blockrun_surf({ path: "..\t/phone/numbers/buy" }) + * blockrun_exa({ path: "..\t/phone/numbers/buy" }) (found on blockrun_surf, + * retired 2026-09-06) * -> guard sees the segment "..\t", not "..", and passes * -> parser strips the tab -> /api/v1/phone/numbers/buy - * -> reserved $0.0095 (surf's price), charged $5.00 + * -> reserved the tool's own price, charged $5.00 * * A 526x under-reserve that also escapes profile scoping (a research-profile * install could buy phone numbers). Verified: all of `..\t/`, `.\t./`, `..\n/`, @@ -45,8 +46,8 @@ export function hasPathTraversal(path: string): boolean { // makes decodeURIComponent throw, the catch falls back to the raw string, and // the later strip leaves the literal segment "%2e%2e" — which is not ".." to // this check, but IS to the parser once it has deleted the same tab. Probed - // live: blockrun_surf path:"%2e%2e/phone/numbers/buy" resolves to - // /api/v1/phone/numbers/buy and quotes $5.001 against surf's $0.0095 reserve, + // live: blockrun_surf (retired 2026-09-06) path:"%2e%2e/phone/numbers/buy" resolved to + // /api/v1/phone/numbers/buy and quoted $5.001 against its $0.0095 reserve, // and escapes profile scoping on the way. Each transformation was tested // alone and passed; only the composition was broken. const asSent = path.replace(/[\t\n\r]/g, ""); @@ -65,10 +66,10 @@ export function hasPathTraversal(path: string): boolean { * per-endpoint price tables key on the bare route, but the gateway router * ignores a trailing `?query`, a trailing slash, or casing when matching — so * classifying the raw slug lets an expensive route (e.g. the $5 - * `phone/numbers/buy`, or a $0.02 surf tier) be mispriced as the cheap default + * `phone/numbers/buy`, or a $0.02 tier) be mispriced as the cheap default * while the gateway still charges full price, defeating the budget pre-check and * under-recording spend. Callers still send the original slug, so a legitimate - * query string (e.g. surf GET params in the path) is preserved. + * query string (e.g. GET params in the path) is preserved. * * CLASSIFY THE ROUTE THAT WILL BE SERVED, NOT THE STRING THE CALLER TYPED. Two * transformations sit between them, and this helper shipped doing neither while diff --git a/src/utils/polymarket/constants.ts b/src/utils/polymarket/constants.ts index a6ab9f9..da95350 100644 --- a/src/utils/polymarket/constants.ts +++ b/src/utils/polymarket/constants.ts @@ -172,6 +172,16 @@ export function getMaxSessionUsd(): number | null { return parseCapEnv("POLYMARKET_MAX_SESSION_USD", process.env.POLYMARKET_MAX_SESSION_USD, null); } +/** + * Optional per-call cap on action:"fund" (Base USDC → own vault), in dollars; + * null = uncapped (unset — fund is a self-to-self move, reversible via + * withdraw, and no cap ever applied, so a default would silently break + * existing users); 0/garbage = freeze. Read per call like the other caps. + */ +export function getMaxFundUsd(): number | null { + return parseCapEnv("POLYMARKET_MAX_FUND_USD", process.env.POLYMARKET_MAX_FUND_USD, null); +} + /** * Bounded pUSD approvals in dollars; null = unlimited (maxUint256). The CTF * ERC-1155 approval is inherently all-or-nothing either way. Only a valid diff --git a/src/utils/polymarket/fund.ts b/src/utils/polymarket/fund.ts index 27ac12a..9a76c54 100644 --- a/src/utils/polymarket/fund.ts +++ b/src/utils/polymarket/fund.ts @@ -12,7 +12,7 @@ import type { Hex } from "viem"; import { BlockrunClient, createPaymentPayload } from "@blockrun/llm"; import { getOrCreateWalletKey, getChainBalance } from "../wallet.js"; import { getPolymarketAccount } from "./client.js"; -import { BASE_CHAIN_ID, BRIDGE_API_HOST, getSigType } from "./constants.js"; +import { BASE_CHAIN_ID, BRIDGE_API_HOST, getMaxFundUsd, getSigType } from "./constants.js"; import { getFundsAddress } from "./positions.js"; import { getPublicClient } from "./setup.js"; import type { ToolResult } from "./orders.js"; @@ -48,6 +48,19 @@ export async function fundVault(input: { amount_usd?: number; confirm?: boolean isError: true, }; } + // Optional per-call ceiling. fund signs an EIP-3009 authorization for the + // FULL amount outside the x402 budget ledger and outside the order caps + // (POLYMARKET_MAX_BET_USD gates executeTrade only). Default: no cap — this is + // the user's own Base USDC into a vault only the same key controls. Checked + // before any RPC/bridge call so the dry-run reports the refusal too. + const maxFund = getMaxFundUsd(); + if (maxFund !== null && input.amount_usd > maxFund) { + return { + text: `Refusing to fund $${input.amount_usd.toFixed(2)}: POLYMARKET_MAX_FUND_USD caps a single funding call at ` + + `$${maxFund.toFixed(2)}. Nothing moved. Fund in smaller calls, or raise/unset the cap if the operator intends it.`, + isError: true, + }; + } const amountUsd = input.amount_usd; let vault: Hex; diff --git a/src/utils/polymarket/orders.ts b/src/utils/polymarket/orders.ts index 18f71a3..b4f0fe4 100644 --- a/src/utils/polymarket/orders.ts +++ b/src/utils/polymarket/orders.ts @@ -192,20 +192,44 @@ function bestQuote(book: OrderBookSummary, side: "buy" | "sell"): number | null return side === "buy" ? Math.min(...prices) : Math.max(...prices); } -function estimateMarketBuyShares(book: OrderBookSummary, amountUsd: number): { shares: number; unfilledUsd: number } { - const asks = (book.asks ?? []) +/** + * Walk the opposite side of the book best-first for a market order. + * buy: `amount` is USD to spend → `filled` is the shares bought. + * sell: `amount` is shares to sell → `filled` is the USD proceeds. + * `worstPrice` is the last level actually consumed — the price the user must + * be shown AND the limit the order must be signed at. Left to itself the SDK + * re-fetches the book at submit time and picks its own limit (the marginal + * level for FOK, the top-of-array level for FAK), so a book that thinned + * between preview and confirm was signed far from the "best ask" the user + * consented to. Books are not guaranteed sorted; sort explicitly. + * Exported for tests. + */ +export function walkBook( + book: OrderBookSummary, + side: "buy" | "sell", + amount: number, +): { filled: number; unfilled: number; worstPrice: number | null } { + const levels = ((side === "buy" ? book.asks : book.bids) ?? []) .map((level) => ({ price: parseFloat(level.price), size: parseFloat(level.size) })) .filter((level) => Number.isFinite(level.price) && level.price > 0 && Number.isFinite(level.size) && level.size > 0) - .sort((a, b) => a.price - b.price); - let remaining = amountUsd; - let shares = 0; - for (const level of asks) { + .sort((a, b) => (side === "buy" ? a.price - b.price : b.price - a.price)); + let remaining = amount; + let filled = 0; + let worstPrice: number | null = null; + for (const level of levels) { if (remaining <= 1e-9) break; - const spend = Math.min(remaining, level.price * level.size); - shares += spend / level.price; - remaining -= spend; + if (side === "buy") { + const spend = Math.min(remaining, level.price * level.size); + filled += spend / level.price; + remaining -= spend; + } else { + const take = Math.min(remaining, level.size); + filled += take * level.price; + remaining -= take; + } + worstPrice = level.price; } - return { shares, unfilledUsd: Math.max(0, remaining) }; + return { filled, unfilled: Math.max(0, remaining), worstPrice }; } /** @@ -403,11 +427,22 @@ export async function executeTrade(input: TradeInput): Promise { }; } + // Market orders: walk the book once, here, and reuse the result for the + // preview, the fillability guards AND the signed limit below. + const walk = !isLimit + ? walkBook(book, input.action, input.action === "buy" ? (input.amount_usd as number) : (size as number)) + : undefined; + // The worst level the walk consumed, rounded conservatively by side (an + // on-grid book level is unchanged), or the best quote if nothing filled. + const worstFillPrice = walk + ? roundToTick(walk.worstPrice ?? (quote as number), tickSize, input.action) + : undefined; + const notional = isLimit ? (price as number) * (size as number) : input.action === "buy" ? (input.amount_usd as number) - : (size as number) * (quote as number); + : (walk as { filled: number }).filled; // walked proceeds, not size × best bid // Enforce the user's hard cap before doing any softer fillability // analysis, so an oversized order always fails for the primary reason. @@ -420,9 +455,7 @@ export async function executeTrade(input: TradeInput): Promise { }; } - const marketBuyEstimate = !isLimit && input.action === "buy" - ? estimateMarketBuyShares(book, input.amount_usd as number) - : undefined; + const marketBuyEstimate = walk && input.action === "buy" ? { shares: walk.filled, unfilledUsd: walk.unfilled } : undefined; const effectiveSize = marketBuyEstimate?.shares ?? size; if (marketBuyEstimate && marketBuyEstimate.unfilledUsd > 0.000001 && (input.order_type ?? "FOK") === "FOK") { @@ -432,6 +465,13 @@ export async function executeTrade(input: TradeInput): Promise { isError: true, }; } + if (walk && input.action === "sell" && walk.unfilled > 0.000001 && (input.order_type ?? "FOK") === "FOK") { + return { + text: `The live bid book cannot fill the full ${size} shares FOK sell ` + + `(about ${walk.unfilled.toFixed(4)} shares have no available bids). Reduce the size or use FAK explicitly.`, + isError: true, + }; + } if (minSize > 0 && (effectiveSize ?? 0) < minSize) { const minimumSpend = input.action === "buy" && quote ? minSize * quote : undefined; @@ -462,8 +502,9 @@ export async function executeTrade(input: TradeInput): Promise { ? ` Limit ${orderKind}: ${size} shares @ ${price} (notional $${notional.toFixed(2)})` : input.action === "buy" ? ` Market ${orderKind}: spend $${(input.amount_usd as number).toFixed(2)}${quote ? ` (best ask ${quote}` + + `, worst fill ≤ ${worstFillPrice}` + `${marketBuyEstimate ? `, est. ${marketBuyEstimate.shares.toFixed(4)} shares` : ""})` : ""}` - : ` Market ${orderKind}: sell ${size} shares${quote ? ` (best bid ${quote}, est. $${notional.toFixed(2)})` : ""}`, + : ` Market ${orderKind}: sell ${size} shares${quote ? ` (best bid ${quote}, worst fill ≥ ${worstFillPrice}, est. $${notional.toFixed(2)})` : ""}`, ` Tick ${tickSize} · negRisk ${negRisk} · min size ${minSize || "n/a"} · fees are taker-only`, ].join("\n"); @@ -490,6 +531,10 @@ export async function executeTrade(input: TradeInput): Promise { outcome: token.outcome, conditionId: token.conditionId, bestQuote: quote, + // Market orders only: the limit the order WILL be signed at (buy: + // max price per share; sell: min price per share). Additive — the + // order card renders it next to the best quote. + worstFillPrice, minSize, maxBetUsd: maxBet, sessionSpentUsd: ledger.totalUsd, @@ -524,6 +569,11 @@ export async function executeTrade(input: TradeInput): Promise { amount: input.action === "buy" ? (input.amount_usd as number) : (size as number), side, orderType: orderKind === "FAK" ? OrderType.FAK : OrderType.FOK, + // The previewed worst fill IS the signed limit (buy: taker + // shares = amount / price; sell: taker USD = shares × price), + // so the exchange can never fill worse than the user saw. With + // a price given the SDK also skips its own second book fetch. + price: worstFillPrice as number, }, options, orderKind === "FAK" ? OrderType.FAK : OrderType.FOK, diff --git a/src/utils/polymarket/redeem.ts b/src/utils/polymarket/redeem.ts index c4abad2..bc87e86 100644 --- a/src/utils/polymarket/redeem.ts +++ b/src/utils/polymarket/redeem.ts @@ -90,15 +90,21 @@ export async function redeemPosition(input: { condition_id?: string; confirm?: b return { text: err instanceof Error ? err.message : String(err), isError: true }; } + // Market metadata (question, outcome tokens, negRisk) via the CLOB. These + // are the ONLY CLOB calls in this function, so they are the only errors + // mapClobError's taxonomy (geoblock 403, creds, "closed" → resolved) can + // legitimately describe; everything below is RPC/relayer and reported raw. + type ClobMarket = { question?: string; neg_risk?: boolean; closed?: boolean; tokens?: ClobMarketToken[] }; + let clob: Awaited>; + let market: ClobMarket; + try { + clob = await getClobClient(); + market = (await clob.getMarket(conditionId)) as ClobMarket; + } catch (err) { + return { text: await mapClobError(err), isError: true }; + } + try { - // Market metadata (question, outcome tokens, negRisk) via the CLOB. - const clob = await getClobClient(); - const market = (await clob.getMarket(conditionId)) as { - question?: string; - neg_risk?: boolean; - closed?: boolean; - tokens?: ClobMarketToken[]; - }; const tokens = (market?.tokens ?? []).filter((t) => t.token_id); if (!tokens.length) return { text: `No tokens found for condition ${conditionId}.`, isError: true }; @@ -291,7 +297,11 @@ export async function redeemPosition(input: { condition_id?: string; confirm?: b }, }; } catch (err) { - const base = await mapClobError(err); + // RPC / relayer / receipt errors — never CLOB — so no mapClobError here: + // an RPC "403" is not a geoblock and a "connection closed" is not a + // resolved market. The raw message is kept verbatim so the relayer's + // deliberate "failed on-chain" / "Do NOT retry" wording reaches the user. + const base = err instanceof Error ? err.message : String(err); // The adapter pulls tokens via safeBatchTransferFrom — a vault set up // before the collateral-adapter approvals were added reverts here. const approvalHint = ` If the transaction reverted, the wallet may be missing the collateral-adapter ` + diff --git a/src/utils/polymarket/relayer.ts b/src/utils/polymarket/relayer.ts index f3851a1..b5f0ee5 100644 --- a/src/utils/polymarket/relayer.ts +++ b/src/utils/polymarket/relayer.ts @@ -148,7 +148,34 @@ export async function sendWalletBatch( opts?: { guidance?: string; trackPendingWithdraw?: boolean }, ): Promise<{ transactionHash?: string }> { const deadlineSec = Math.floor(Date.now() / 1000) + BATCH_DEADLINE_SECS; - const response = await (await getRelayClient()).executeDepositWalletBatch(calls, depositWallet, String(deadlineSec)); + let response: Awaited>; + try { + response = await (await getRelayClient()).executeDepositWalletBatch(calls, depositWallet, String(deadlineSec)); + } catch (err) { + // The SDK signs, THEN posts. A lost response (proxy 502/504, reset — the + // SDK surfaces these as `{"error":"connection error"}` or a 5xx "request + // error") leaves the signature executable until deadlineSec with nothing + // on disk to say so, and the old error text invited an immediate retry — + // the #72.1 double-send on the submit side. Only a definite 4xx proves the + // relayer accepted nothing. Pre-sign failures (signer/config/nonce) are + // 4xx-free too but cannot have signed anything; we cannot tell them apart + // from a lost post here, so the conservative side wins for tracked + // (money-moving) batches: arm the guard, say so, and let the deadline + // clear it (withdraw.ts). Untracked batches (approvals, wrap) are safe to + // retry and rethrow as before. + const message = err instanceof Error ? err.message : String(err); + const definitelyRejected = /"status":4\d\d/.test(message); + if (opts?.trackPendingWithdraw && !definitelyRejected) { + saveState({ pendingWithdraw: { transactionID: "unknown", deadline: deadlineSec } }); + throw new Error( + `${description}: the relayer returned no transaction id (${message}). It may still have ACCEPTED the ` + + `signed batch — a transfer may already be in flight, and the signature stays executable for ` + + `${BATCH_DEADLINE_SECS / 60} minutes. Do NOT retry yet: wait for that deadline to pass, then ` + + `${opts?.guidance ?? 're-run action:"setup" to re-check state'}.`, + ); + } + throw err; + } if (opts?.trackPendingWithdraw) { saveState({ pendingWithdraw: { transactionID: response.transactionID, deadline: deadlineSec } }); } diff --git a/src/utils/polymarket/setup.ts b/src/utils/polymarket/setup.ts index a1b9126..85630eb 100644 --- a/src/utils/polymarket/setup.ts +++ b/src/utils/polymarket/setup.ts @@ -257,12 +257,26 @@ async function runSetupDepositWallet(opts: { confirm: boolean }): Promise<{ text const account = getPolymarketAccount(); // 1. Derive (pure CREATE2 math) + persist, keyed to the current signer. + // `deployed`/`approvalsDone` describe ONE vault. saveState is a shallow + // merge, so after a signer rotation (loadDepositWalletForSigner refuses + // the old vault → a new CREATE2 address is derived) the old vault's + // deployed:true used to survive, short-circuit the deploy step below, and + // have setup print "✅ deployed" + "bridge USDC here" for a vault with no + // code — which the bridge sweeps and never delivers (fund.ts). Snapshot + // BEFORE the merge and reset the per-wallet flags when the address moves. + const prev = loadState(); const depositWallet = (loadDepositWalletForSigner(account.address) as Hex | undefined) ?? (await deriveDepositWallet()); - saveState({ depositWallet, signer: account.address }); + const sameWallet = prev.depositWallet?.toLowerCase() === depositWallet.toLowerCase(); + saveState( + sameWallet + ? { depositWallet, signer: account.address } + : { depositWallet, signer: account.address, deployed: false, approvalsDone: false }, + ); // 2. Deploy if missing — gasless, moves no funds, ownership is baked into - // the CREATE2 address, so no confirm gate is needed here. - let deployed = loadState().deployed === true || (await isDepositWalletDeployed(depositWallet)); + // the CREATE2 address, so no confirm gate is needed here. The persisted + // flag is only trusted for the wallet it was written for. + let deployed = (sameWallet && prev.deployed === true) || (await isDepositWalletDeployed(depositWallet)); let deployTxHash: string | undefined; if (!deployed) { const res = await deployDepositWallet(); diff --git a/src/utils/polymarket/withdraw.ts b/src/utils/polymarket/withdraw.ts index 10402cf..cc74ea9 100644 --- a/src/utils/polymarket/withdraw.ts +++ b/src/utils/polymarket/withdraw.ts @@ -19,7 +19,7 @@ // wrapped to pUSD through the collateral onramp first (sweep design from // @KillerQueen-Z's #59/#66, tracked in #71). import axios from "axios"; -import { encodeFunctionData, formatUnits, http, createWalletClient, type Hex } from "viem"; +import { encodeFunctionData, formatUnits, http, createWalletClient, isAddress, type Hex } from "viem"; import { polygon } from "viem/chains"; import { BASE_CHAIN_ID, @@ -37,7 +37,6 @@ import { import { getPolymarketAccount } from "./client.js"; import { assertTransactionSucceeded } from "./transactions.js"; import type { ToolResult } from "./orders.js"; -import { mapClobError } from "./orders.js"; import { getFundsAddress } from "./positions.js"; import { loadState, saveState } from "./creds.js"; import { getRelayerTransactionState, sendWalletBatch } from "./relayer.js"; @@ -134,7 +133,24 @@ export async function withdrawFunds(input: WithdrawInput): Promise { } catch (err) { return { text: err instanceof Error ? err.message : String(err), isError: true }; } - const recipient = (input.to_address as Hex) || getPolymarketAccount().address; + // The destination is forwarded to the bridge as `recipientAddr` and the USDC + // lands wherever it says — irreversibly. Validate BEFORE any I/O: strict + // isAddress rejects non-addresses and mixed-case strings whose EIP-55 + // checksum does not match (the transposition-typo shape); all-lowercase input + // carries no checksum and is accepted as-is. + if (input.to_address !== undefined && !isAddress(input.to_address, { strict: true })) { + return { + text: `to_address must be a valid 0x… Base address (40 hex chars; if mixed-case, the checksum must match). ` + + `Got ${JSON.stringify(input.to_address)}. Nothing withdrawn.`, + isError: true, + }; + } + const agent = getPolymarketAccount().address; + const recipient = (input.to_address as Hex | undefined) ?? agent; + // The dry-run is the one human checkpoint before the transfer. Label the + // destination from what it IS, not from where the default would have gone: + // a caller-supplied third-party address used to print as "agent wallet". + const isCustom = recipient.toLowerCase() !== agent.toLowerCase(); try { // Refuse to sign while an earlier withdrawal batch may still land: its @@ -145,13 +161,18 @@ export async function withdrawFunds(input: WithdrawInput): Promise { if (pending && input.confirm === true) { const graceSec = 60; // relayer can mine right at the deadline; don't race it if (Math.floor(Date.now() / 1000) < pending.deadline + graceSec) { - const state = await getRelayerTransactionState(pending.transactionID); + // "unknown" = the relayer never returned an id (submit response lost, + // see relayer.ts sendWalletBatch). There is nothing to look up, and the + // signed batch may still land — block until the deadline passes. + const idUnknown = pending.transactionID === "unknown"; + const state = idUnknown ? undefined : await getRelayerTransactionState(pending.transactionID); if (state === "STATE_MINED" || state === "STATE_CONFIRMED" || state === "STATE_FAILED" || state === "STATE_INVALID") { saveState({ pendingWithdraw: undefined }); } else { const waitSecs = pending.deadline + graceSec - Math.floor(Date.now() / 1000); + const stateLabel = idUnknown ? "unknown — the submit response was lost" : (state ?? "unreachable"); return { - text: `A previous withdrawal (relayer tx ${pending.transactionID}, state: ${state ?? "unreachable"}) ` + + text: `A previous withdrawal (relayer tx ${pending.transactionID}, state: ${stateLabel}) ` + `may still execute — its signed transfer stays valid for up to ~${waitSecs}s more. Signing another ` + `one now could double-send. Re-run after that window, when the balance reads will show what happened.`, isError: true, @@ -188,7 +209,7 @@ export async function withdrawFunds(input: WithdrawInput): Promise { `DRY RUN — nothing withdrawn.`, `Withdraw $${amountUsd.toFixed(2)} → native USDC on Base`, ` from deposit wallet: ${owner}`, - ` to (agent wallet): ${recipient}`, + ` to: ${recipient}${isCustom ? " ⚠️ CUSTOM destination — NOT your agent wallet" : " (your agent wallet)"}`, ...(wrapRaw > 0n ? [``, ` First wrap: $${Number(formatUnits(wrapRaw, PUSD_DECIMALS)).toFixed(2)} legacy USDC.e → pUSD (collateral onramp, same wallet)`] : []), @@ -275,7 +296,7 @@ export async function withdrawFunds(input: WithdrawInput): Promise { text: [ `✅ Withdrawal submitted: $${amountUsd.toFixed(2)} → USDC on Base`, ...(wrapRaw > 0n ? [` (included wrapping $${Number(formatUnits(wrapRaw, PUSD_DECIMALS)).toFixed(2)} legacy USDC.e → pUSD first)`] : []), - ` to your agent wallet: ${recipient}`, + ` to ${isCustom ? "CUSTOM address (not your agent wallet)" : "your agent wallet"}: ${recipient}`, ...(txHash ? [` pUSD transfer tx: https://polygonscan.com/tx/${txHash}`] : []), ` The bridge unwraps + delivers USDC to Base (usually within a minute).`, ` Track: GET ${BRIDGE_API_HOST}/status/${owner}`, @@ -287,6 +308,27 @@ export async function withdrawFunds(input: WithdrawInput): Promise { }, }; } catch (err) { - return { text: await mapClobError(err), isError: true }; + // No CLOB call happens anywhere in this function, so mapClobError's + // taxonomy does not apply: a bridge 403 used to come back as "point + // POLYMARKET_CLOB_HOST + POLYMARKET_RELAYER_URL at a permitted-region + // relay" (the relay does not serve the bridge) and any transport message + // containing "closed" as "market resolved, go redeem". Report the bridge + // as the bridge; pass everything else through verbatim — the relayer's + // anti-retry wording (sendWalletBatch) must reach the user unchanged. + return { text: describeWithdrawError(err), isError: true }; + } +} + +/** Plain, source-honest error text for the withdraw path. Exported for tests. */ +export function describeWithdrawError(err: unknown): string { + const message = err instanceof Error ? err.message : String(err); + const e = err as { isAxiosError?: boolean; response?: { status?: number; data?: unknown } }; + if (e?.isAxiosError === true) { + const status = e.response?.status; + const data = e.response?.data !== undefined ? ` — ${typeof e.response.data === "string" ? e.response.data : JSON.stringify(e.response.data)}` : ""; + return `Polymarket bridge request failed (POST ${BRIDGE_API_HOST}/withdraw${status ? `, HTTP ${status}` : ""}): ${message}${data}. ` + + `Nothing was transferred — the pUSD move only happens after the bridge answers. This is the BRIDGE host ` + + `(POLYMARKET_BRIDGE_HOST), not the CLOB/relayer egress; check the bridge status, then retry.`; } + return message; } diff --git a/src/utils/solana-402.ts b/src/utils/solana-402.ts index e0c1f63..2213cac 100644 --- a/src/utils/solana-402.ts +++ b/src/utils/solana-402.ts @@ -109,8 +109,12 @@ export interface SolanaPaidAsyncPostOptions { resignIntervalMs?: number; /** Maximum reactive re-signs after a completed poll rejects a stale signature. Defaults to SOLANA_ASYNC_MAX_REACTIVE_RESIGNS. */ maxReactiveResigns?: number; - /** Called after the authoritative quote is parsed and before anything is signed. */ - onQuote?: (quotedUsd: number | null) => void; + /** + * Called after the authoritative quote is parsed and before anything is + * signed. `details` is the decoded 402 (amount, recipient, resource + * description) so a caller can check WHAT was quoted, not just how much. + */ + onQuote?: (quotedUsd: number | null, details: ReturnType) => void; } type SolanaPaymentContext = { @@ -190,7 +194,7 @@ export async function solanaPaidPost( * without paying — e.g. to re-check the real price against a budget cap when * the Solana gateway's marked-up amount exceeds the caller's estimate. */ - onQuote?: (quotedUsd: number | null) => void; + onQuote?: (quotedUsd: number | null, details: ReturnType) => void; }, ): Promise { // resolveSolanaKey, not the SDK's file-only loader: under @@ -225,7 +229,7 @@ export async function solanaPaidPost( // Hand the caller the REAL quoted price before we sign/pay, so it can re-check // the marked-up Solana amount against its budget cap and abort (by throwing) // if it would overshoot — the amount is only known now, after the quote. - opts?.onQuote?.(context.paidUsd); + opts?.onQuote?.(context.paidUsd, context.details); const paymentPayload = await signSolanaChallenge(context, url, privateKey); // Step 2: paid request. The signed SPL transaction embeds a recent blockhash @@ -308,7 +312,7 @@ export async function solanaPaidAsyncPost( if (original.paidUsd === null) { throw new PaymentError(`The gateway's Solana quote carried an unreadable amount (${JSON.stringify(original.details.amount)}); refusing to sign it. No charge was made.`); } - opts.onQuote?.(original.paidUsd); + opts.onQuote?.(original.paidUsd, original.details); // Stamp BEFORE signing: the blockhash is fetched inside the sign call, and a // slow submit afterwards must not make the tracked age lag the real one. diff --git a/src/utils/wallet.ts b/src/utils/wallet.ts index 960f2c2..a086a13 100644 --- a/src/utils/wallet.ts +++ b/src/utils/wallet.ts @@ -9,8 +9,11 @@ import { SolanaLLMClient, AnthropicClient, getOrCreateWallet, - getOrCreateSolanaWallet, + createSolanaWallet, + saveSolanaWallet, + solanaPublicKey, loadSolanaWallet, + USDC_SOLANA, getPaymentLinks, formatWalletCreatedMessage, formatNeedsFundingMessage, @@ -246,10 +249,14 @@ export async function ensureBothWallets(): Promise<{ const chainBefore = readChainPreference() === null ? getChain() : null; const evm = ensureEvmWallet(); - const sol = await getOrCreateSolanaWallet(); - if (sol.isNew) { - console.error(formatWalletCreatedMessage(sol.address)); - } + // NOT the SDK's getOrCreateSolanaWallet(): that loader knows only the env var + // and the file. Under BLOCKRUN_KEYCHAIN=strict the file is retired once the + // key is in the keychain, so the SDK saw an empty slate, minted keypair B, + // wrote it to .solana-session — and the next resolveSolanaKey() mirrored B + // into the keychain with -U, over the funded key A, then deleted the file. + // A was then nowhere. ensureSolanaWallet() reads the keychain first and + // refuses to mint when the keychain could not be read (audit 2026-09-08). + const sol = await ensureSolanaWallet(); if (chainBefore !== null && getChain() !== chainBefore) { // writeAutoChain, NOT setChain: this is the machine preserving continuity, @@ -433,10 +440,20 @@ export function getOrCreateWalletKey(): `0x${string}` { return info.privateKey as `0x${string}`; } -// Resolved once per process. buildSolanaClient() is called per-request on the -// non-cached paths (blockrun_chat, modal), and a keychain read spawns a -// subprocess — fine once, not fine on every paid call. -let _solanaKey: string | null | undefined; +// Resolved once per process on a HIT. buildSolanaClient() is called +// per-request on the non-cached paths (blockrun_chat, modal), and a keychain +// read spawns a subprocess — fine once, not fine on every paid call. A MISS is +// deliberately not memoised: a wallet provisioned later in the same process (by +// ensureSolanaWallet, or by another process such as the CLI) must become +// visible without a restart — the old `null` cache made the first status call +// of a fresh install poison every later call. +let _solanaKey: string | undefined; + +type SolanaKeyResolution = { + key?: string; + /** Set when the keychain was consulted and the read FAILED (not "absent"). */ + keychainError?: string; +}; /** * Solana key precedence, mirroring the EVM path: @@ -445,30 +462,79 @@ let _solanaKey: string | null | undefined; * file is not silently undone by a stale keychain entry; the keychain carries * the key only once the file is gone (strict mode). A key found in the file is * mirrored into the keychain on the way past. + * + * Uses keychainRead, not keychainLoad: "absent" and "error" must stay apart, + * because ensureSolanaWallet() decides whether to CREATE a wallet on the + * difference — the exact rule keychain.ts states for the EVM path. */ -export function resolveSolanaKey(): string | undefined { - if (process.env.SOLANA_WALLET_KEY) return process.env.SOLANA_WALLET_KEY; - if (_solanaKey !== undefined) return _solanaKey ?? undefined; +function resolveSolanaKeyDetailed(): SolanaKeyResolution { + if (process.env.SOLANA_WALLET_KEY) return { key: process.env.SOLANA_WALLET_KEY }; + if (_solanaKey) return { key: _solanaKey }; + let keychainError: string | undefined; // Same precedence correction as the EVM path: an existing .solana-session is // the user's current intent, so it outranks whatever the keychain remembers. if (getKeychainMode() !== "off" && !fs.existsSync(SOLANA_WALLET_FILE_PATH)) { - const stored = keychainLoad(SOLANA_KEY_ACCOUNT); - if (stored) { - _solanaKey = stored; - return stored; + const read = keychainRead(SOLANA_KEY_ACCOUNT); + if (read.status === "found") { + _solanaKey = read.value; + return { key: read.value }; } + if (read.status === "error") keychainError = read.detail; } const fromFile = loadSolanaWallet(); - if (fromFile) persistKey(SOLANA_KEY_ACCOUNT, fromFile, SOLANA_WALLET_FILE_PATH); - _solanaKey = fromFile ?? null; - return fromFile ?? undefined; + if (fromFile) { + persistKey(SOLANA_KEY_ACCOUNT, fromFile, SOLANA_WALLET_FILE_PATH); + _solanaKey = fromFile; + return { key: fromFile }; + } + return { keychainError }; } -/** Drop the cached Solana key. Test seam, and used when the wallet is re-provisioned. */ +export function resolveSolanaKey(): string | undefined { + return resolveSolanaKeyDetailed().key; +} + +let _solanaWalletInfo: { address: string; privateKey: string; isNew: boolean } | null = null; + +/** + * The Solana twin of ensureEvmWallet(): return the existing wallet from + * whichever store holds it, and mint one ONLY when every store says "absent". + * A keychain read that FAILED is not a read that found nothing — the file is + * already gone in strict mode, so minting here would orphan a funded key that + * is very likely still sitting in a keychain we merely could not open. + */ +export async function ensureSolanaWallet(): Promise<{ address: string; privateKey: string; isNew: boolean }> { + if (_solanaWalletInfo) return _solanaWalletInfo; + const { key, keychainError } = resolveSolanaKeyDetailed(); + if (key) { + _solanaWalletInfo = { address: await solanaPublicKey(key), privateKey: key, isNew: false }; + return _solanaWalletInfo; + } + if (keychainError !== undefined) { + throw new Error( + `Could not read the Solana wallet key from the OS keychain (${keychainError}), and ` + + `~/.blockrun/.solana-session does not exist (BLOCKRUN_KEYCHAIN=strict retires it once the key is in the keychain). ` + + `Refusing to create a new Solana wallet — your existing one is most likely still in the keychain. ` + + `Unlock the keychain and retry, or set SOLANA_WALLET_KEY to your key. Nothing was charged.`, + ); + } + const created = await createSolanaWallet(); + saveSolanaWallet(created.privateKey); + // Mirror into the keychain; strict mode then retires the file after a + // verified read-back — the same sequence ensureEvmWallet() runs. + persistKey(SOLANA_KEY_ACCOUNT, created.privateKey, SOLANA_WALLET_FILE_PATH); + _solanaKey = created.privateKey; + _solanaWalletInfo = { address: created.address, privateKey: created.privateKey, isNew: true }; + console.error(formatWalletCreatedMessage(created.address)); + return _solanaWalletInfo; +} + +/** Drop the cached Solana key and wallet. Test seam, and used when the wallet is re-provisioned. */ export function resetSolanaKeyCache(): void { _solanaKey = undefined; + _solanaWalletInfo = null; } /** @@ -502,8 +568,18 @@ function buildSolanaClient(timeout?: number): SolanaLLMClient { return new SolanaLLMClient({ apiKey, ...(timeout ? { timeout } : {}) }); } const privateKey = resolveSolanaKey(); - const opts = { ...(privateKey ? { privateKey } : {}), ...(timeout ? { timeout } : {}) }; - return new SolanaLLMClient(Object.keys(opts).length ? opts : undefined); + if (!privateKey) { + // The SDK constructor would throw "Private key required. Pass privateKey in + // options or set SOLANA_WALLET_KEY" — true, and useless to someone on a + // fresh install where Solana is the default chain and nothing has minted a + // wallet yet. Provisioning is async (ensureSolanaWallet) and this factory + // is sync, so name the remedy instead of the symptom. + throw new Error( + `No Solana wallet on this machine yet. Run blockrun_wallet action:"setup" (or action:"chain" chain:"solana") to create one, ` + + `or set SOLANA_WALLET_KEY. Nothing was charged.`, + ); + } + return new SolanaLLMClient({ privateKey, ...(timeout ? { timeout } : {}) }); } export function getClient(): ApiClient { @@ -626,15 +702,18 @@ export async function getWalletInfo(): Promise { }; } if (getChain() === "solana") { - const client = getClient() as SolanaLLMClient; - const address = await client.getWalletAddress(); + // ensureSolanaWallet, not getClient(): on a fresh install (Solana is the + // default since 0.46.0) nothing had minted a wallet yet, so every + // status/setup/qr/deposit call died in the SDK constructor before the + // one action that creates wallets was reached. Mirrors the EVM branch. + const info = await ensureSolanaWallet(); return { - address, + address: info.address, network: "Solana" as const, chainId: null as number | null, currency: "USDC", - isNew: false, - explorerUrl: `https://solscan.io/account/${address}`, + isNew: info.isNew, + explorerUrl: `https://solscan.io/account/${info.address}`, fundingUrl: "https://sol.blockrun.ai", }; } @@ -653,9 +732,40 @@ export async function getWalletInfo(): Promise { export { formatNeedsFundingMessage }; -async function getSolanaUsdcBalance(): Promise { +const DEFAULT_SOLANA_RPC_URL = "https://sol.blockrun.ai/api/v1/solana/rpc"; + +/** + * USDC balance of an explicit Solana ADDRESS. The SDK's getBalance() only ever + * reads the client's own wallet and ignores which address the caller is + * displaying, so the status screen could print address B beside the balance + * of key A. Same RPC call the SDK makes, keyed on the address we show. Returns + * null (not 0) when the RPC cannot be reached — "unavailable" is honest, + * "$0.00" beside a funded address is not. + */ +async function getSolanaUsdcBalance(address: string): Promise { + const rpcUrl = process.env.SOLANA_RPC_URL || DEFAULT_SOLANA_RPC_URL; try { - return await buildSolanaClient().getBalance(); + const response = await fetch(rpcUrl, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + jsonrpc: "2.0", + id: 1, + method: "getTokenAccountsByOwner", + params: [address, { mint: USDC_SOLANA }, { encoding: "jsonParsed" }], + }), + signal: AbortSignal.timeout(8000), + }); + const data = await response.json() as { + result?: { value?: Array<{ account?: { data?: { parsed?: { info?: { tokenAmount?: { uiAmount?: number } } } } } }> }; + error?: unknown; + }; + if (data.error || !data.result) return null; + let total = 0; + for (const acct of data.result.value ?? []) { + total += acct.account?.data?.parsed?.info?.tokenAmount?.uiAmount ?? 0; + } + return total; } catch { return null; } } @@ -696,7 +806,7 @@ async function getBaseUsdcBalance(address: string): Promise { /** USDC balance for an explicit chain — used to show BOTH wallets at once. */ export async function getChainBalance(chain: "base" | "solana", address: string): Promise { - return chain === "solana" ? getSolanaUsdcBalance() : getBaseUsdcBalance(address); + return chain === "solana" ? getSolanaUsdcBalance(address) : getBaseUsdcBalance(address); } export async function getUsdcBalance(address: string): Promise { diff --git a/test/account-async-billed.test.ts b/test/account-async-billed.test.ts new file mode 100644 index 0000000..5c596f0 --- /dev/null +++ b/test/account-async-billed.test.ts @@ -0,0 +1,181 @@ +// Run with: npm test (tsx --experimental-test-module-mocks --test) +// +// blockrun_video and blockrun_music on the ACCOUNT rail, driven through the real +// handlers with the HTTP layer scripted. The account rail bills an async job at +// submit, so when the job then times out or fails the tool has two duties the +// wallet rails do not: book the charge in the local ledger (the reservation is +// released in finally, so without a booking the cap silently rises by the whole +// clip price), and never tell the caller to "try again" — that submits and bills +// a second job. Both were missing; isTimeoutError matched the deadline message +// and glued "please try again" onto a note saying the job was billed. +import { test, mock, beforeEach, afterEach } from "node:test"; +import assert from "node:assert/strict"; +import type { BudgetState } from "../src/types.js"; + +process.env.BLOCKRUN_API_KEY = "brk_live_testkeyfortestsonly0000"; + +function headers(map: Record = {}) { + const lower = Object.fromEntries(Object.entries(map).map(([k, v]) => [k.toLowerCase(), v])); + return { get: (name: string) => lower[name.toLowerCase()] ?? null }; +} + +let script: Array<() => unknown> = []; +let fetchCalls = 0; +mock.module("../src/utils/http.js", { + namedExports: { + fetchWithTimeout: async () => { + fetchCalls++; + const next = script.shift(); + if (!next) throw new Error("UNEXPECTED_NETWORK_CALL"); + return next(); + }, + // The real predicate, restated: the tools must classify BilledJobError + // BEFORE this matches "did not complete within". + isTimeoutError: (err: unknown) => { + const name = err instanceof Error ? err.name : ""; + if (name === "AbortError" || name === "TimeoutError") return true; + const msg = (err instanceof Error ? err.message : String(err)).toLowerCase(); + return msg.includes("abort") || msg.includes("timeout") || msg.includes("timed out") || msg.includes("did not complete within"); + }, + }, +}); +mock.module("../src/utils/wallet.js", { + namedExports: { + getApiBase: () => "https://api.blockrun.ai", + resolveGatewayUrl: (u: string) => (u.startsWith("http") ? u : `https://api.blockrun.ai${u.startsWith("/api/") ? u.slice(4) : u}`), + getChain: () => "solana", // the account rail must win over the chain + getOrCreateWalletKey: () => { throw new Error("account rail must not touch a wallet key"); }, + getWalletInfo: async () => ({ address: "0xTEST" }), + resolveSolanaKey: () => undefined, + }, +}); +mock.module("../src/utils/ssrf.js", { + namedExports: { isBlockedFetchHostResolved: async () => false, isBlockedFetchHost: () => false }, +}); +mock.module("@blockrun/llm", { + namedExports: { + createPaymentPayload: async () => { throw new Error("account rail must not sign a payment"); }, + parsePaymentRequired: () => ({}), + extractPaymentDetails: () => ({}), + }, +}); + +const { registerVideoTool } = await import("../src/tools/video.js"); +const { registerMusicTool } = await import("../src/tools/music.js"); +const { withTxFee } = await import("../src/utils/tx-fee.js"); +const MUSIC_COST = withTxFee(0.1575); + +// A movable clock: jumping it past the poll deadline inside the submit response +// skips the while-loop entirely, so the deadline path runs with no 5s sleep. +const realNow = Date.now; +let clockOffset = 0; +mock.method(Date, "now", () => realNow() + clockOffset); + +function makeHarness(register: (server: any, budget: BudgetState) => void) { + let handler: ((args: Record) => Promise) | undefined; + const server = { + registerTool: (_n: string, _c: unknown, h: any) => { handler = h; }, + server: { getClientCapabilities: () => ({}) }, + } as any; + const budget: BudgetState = { limit: null, spent: 0, calls: 0, agents: new Map() }; + register(server, budget); + return { call: (args: Record) => handler!(args), budget }; +} +const text = (res: any) => res.content.map((c: any) => c.text).join("\n"); + +const submit202 = (cost: string | null, pollUrl: string, jump: boolean) => () => { + if (jump) clockOffset += 3_600_000; + return { + status: 202, ok: true, + headers: headers(cost === null ? {} : { "x-blockrun-cost-usd": cost }), + json: async () => ({ id: "job_1", poll_url: pollUrl, status: "queued" }), + }; +}; +const abortError = () => { const e = new Error("This operation was aborted"); e.name = "AbortError"; return e; }; + +beforeEach(() => { script = []; fetchCalls = 0; clockOffset = 0; }); +afterEach(() => { assert.equal(script.length, 0, "unconsumed scripted responses"); }); + +// --------------------------------------------------------------------------- +// video +// --------------------------------------------------------------------------- + +test("video: a job that times out after submit is BOOKED at the settled cost and never told to retry", async () => { + script = [submit202("9.450000", "/api/v1/videos/generations/job_1", true)]; + const { call, budget } = makeHarness(registerVideoTool); + const res = await call({ prompt: "a cube", model: "bytedance/seedance-2.5", duration_seconds: 30 }); + const t = text(res); + assert.equal(res.isError, true, t); + assert.match(t, /billed to the BlockRun account/); + assert.match(t, /job_1/); + assert.match(t, /dashboard\/activity/); + assert.match(t, /bills a second job/); + assert.doesNotMatch(t, /please try again/); + assert.doesNotMatch(t, /try again/i); + assert.ok(Math.abs(budget.spent - 9.45) < 1e-9, `ledger must carry the submit charge: spent=${budget.spent}`); + assert.equal(fetchCalls, 1); +}); + +test("video: a submit that gets no answer books the estimate and says the job MAY have been billed", async () => { + script = [() => { throw abortError(); }]; + const { call, budget } = makeHarness(registerVideoTool); + const res = await call({ prompt: "a cube", model: "xai/grok-imagine-video" }); + const t = text(res); + assert.equal(res.isError, true, t); + assert.match(t, /MAY have been accepted and billed/); + assert.match(t, /dashboard\/activity/); + assert.doesNotMatch(t, /try again/i); + assert.doesNotMatch(t, /No payment was taken|not charged|nothing was charged/i, "a lost submit is not a known refund"); + // Conservative: the estimate is booked so the cap cannot under-count a real charge. + assert.ok(budget.spent > 0.39 && budget.spent < 0.41, `estimate booked: spent=${budget.spent}`); +}); + +test("video: the happy path books the settled submit cost once", async () => { + script = [ + submit202("0.401000", "/api/v1/videos/generations/job_1", false), + () => ({ status: 200, ok: true, headers: headers(), json: async () => ({ status: "completed", data: [{ url: "https://blockrun.ai/media/job_1.mp4", duration_seconds: 8 }] }) }), + ]; + const { call, budget } = makeHarness(registerVideoTool); + const res = await call({ prompt: "a cube", model: "xai/grok-imagine-video" }); + assert.notEqual(res.isError, true, text(res)); + assert.equal(res.structuredContent.cost_usd, 0.401); + assert.ok(Math.abs(budget.spent - 0.401) < 1e-9, `booked once: spent=${budget.spent}`); +}); + +// --------------------------------------------------------------------------- +// music +// --------------------------------------------------------------------------- + +test("music: a job that times out after submit books the estimate when no cost header came back", async () => { + script = [submit202(null, "/api/v1/audio/generations/job_1", true)]; + const { call, budget } = makeHarness(registerMusicTool); + const res = await call({ prompt: "lofi beat" }); + const t = text(res); + assert.equal(res.isError, true, t); + assert.match(t, /billed to the BlockRun account/); + assert.match(t, /job_1/); + assert.match(t, /dashboard\/activity/); + assert.doesNotMatch(t, /try again/i); + assert.doesNotMatch(t, /peak load/); + assert.ok(Math.abs(budget.spent - MUSIC_COST) < 1e-9, `estimate booked for a billed job: spent=${budget.spent}`); +}); + +test("music: a job that times out after submit books the settled cost when the header is present", async () => { + script = [submit202("0.157500", "/api/v1/audio/generations/job_1", true)]; + const { call, budget } = makeHarness(registerMusicTool); + const res = await call({ prompt: "lofi beat" }); + assert.equal(res.isError, true, text(res)); + assert.ok(Math.abs(budget.spent - 0.1575) < 1e-9, `settled cost booked: spent=${budget.spent}`); +}); + +test("music: a submit that gets no answer books the estimate and says the job MAY have been billed", async () => { + script = [() => { throw abortError(); }]; + const { call, budget } = makeHarness(registerMusicTool); + const res = await call({ prompt: "lofi beat" }); + const t = text(res); + assert.equal(res.isError, true, t); + assert.match(t, /MAY have been accepted and billed/); + assert.doesNotMatch(t, /try again/i); + assert.doesNotMatch(t, /peak load/); + assert.ok(Math.abs(budget.spent - MUSIC_COST) < 1e-9, `estimate booked: spent=${budget.spent}`); +}); diff --git a/test/api-key-call-async.test.ts b/test/api-key-call-async.test.ts new file mode 100644 index 0000000..3e7cd8a --- /dev/null +++ b/test/api-key-call-async.test.ts @@ -0,0 +1,212 @@ +// Run with: npm test (tsx --experimental-test-module-mocks --test) +// +// Drives apiKeyAsyncPost against a scripted fetch. This is the one rail where the +// money is gone at SUBMIT — the gateway bills an async media job the moment it +// answers 202, and the polls are free — so every exit after a successful submit +// has to carry two things the wallet rails never need: the fact that the job is +// already billed (with its id), and the cost the submit response reported, so +// the caller can book it. Before this file existed nothing drove the function; +// image-account-cost.test.ts mocks the module out entirely. +import { test, mock, beforeEach, afterEach } from "node:test"; +import assert from "node:assert/strict"; + +process.env.BLOCKRUN_API_KEY = "brk_live_testkeyfortestsonly0000"; + +function headers(map: Record = {}) { + const lower = Object.fromEntries(Object.entries(map).map(([k, v]) => [k.toLowerCase(), v])); + return { get: (name: string) => lower[name.toLowerCase()] ?? null }; +} + +type Scripted = { url: string; method: string; timeoutMs: number }; +let script: Array<() => unknown> = []; +let requests: Scripted[] = []; +mock.module("../src/utils/http.js", { + namedExports: { + fetchWithTimeout: async (url: string, init: { method?: string }, timeoutMs: number) => { + requests.push({ url, method: init.method || "GET", timeoutMs }); + const next = script.shift(); + if (!next) throw new Error("UNEXPECTED_NETWORK_CALL"); + return next(); + }, + }, +}); +mock.module("../src/utils/wallet.js", { + namedExports: { + getApiBase: () => "https://api.blockrun.ai", + resolveGatewayUrl: (u: string) => (u.startsWith("http") ? u : `https://api.blockrun.ai${u.startsWith("/api/") ? u.slice(4) : u}`), + }, +}); + +const { apiKeyAsyncPost, BilledJobError } = await import("../src/utils/api-key-call.js"); + +beforeEach(() => { script = []; requests = []; }); +afterEach(() => { assert.equal(script.length, 0, "unconsumed scripted responses"); }); + +const POLL = "/api/v1/videos/generations/job_1"; +const submit = (cost: string | null = "9.450000") => ({ + status: 202, ok: true, + headers: headers(cost === null ? {} : { "x-blockrun-cost-usd": cost }), + json: async () => ({ id: "job_1", poll_url: POLL, status: "queued" }), +}); +const poll = (status: string, extra: Record = {}, http = status === "completed" ? 200 : 202) => ({ + status: http, ok: http >= 200 && http < 300, headers: headers(), + json: async () => ({ status, ...extra }), +}); +const statusOnly = (http: number, hdrs: Record = {}, body: Record = {}) => ({ + status: http, ok: false, headers: headers(hdrs), json: async () => body, +}); +const abortError = () => { const e = new Error("This operation was aborted"); e.name = "AbortError"; return e; }; +const fast = { pollBudgetMs: 10_000, pollIntervalMs: 1 }; +const gets = () => requests.filter((r) => r.method === "GET"); + +// --------------------------------------------------------------------------- +// Transient poll trouble keeps polling — the job is paid for. +// --------------------------------------------------------------------------- + +test("a poll fetch that rejects is retried inside the deadline, not fatal", async () => { + script = [ + submit, + () => { throw new TypeError("fetch failed"); }, + () => { throw abortError(); }, + () => poll("in_progress"), + () => poll("completed", { data: [{ url: "https://blockrun.ai/media/job_1.mp4" }] }), + ]; + const result = await apiKeyAsyncPost("/v1/videos/generations", { prompt: "t" }, fast); + assert.equal(result.jobId, "job_1"); + assert.equal(result.paidUsd, 9.45, "the cost rides the submit response"); + assert.equal((result.data.data as Array<{ url: string }>)[0].url, "https://blockrun.ai/media/job_1.mp4"); + assert.equal(gets().length, 4); + assert.equal(gets()[0].url, `https://api.blockrun.ai/v1/videos/generations/job_1`, "poll URL resolves onto the account API, not the wallet gateway"); +}); + +test("transient proxy statuses on a poll (429/502/503/504/522/524) keep polling", async () => { + script = [ + submit, + () => statusOnly(503), + () => statusOnly(429, { "retry-after": "0" }), + () => statusOnly(502), + () => statusOnly(504), + () => statusOnly(522), + () => statusOnly(524), + () => poll("completed", { data: [{ url: "u" }] }), + ]; + const result = await apiKeyAsyncPost("/v1/audio/generations", { prompt: "t" }, fast); + assert.equal(result.jobId, "job_1"); + assert.equal(gets().length, 7); +}); + +// --------------------------------------------------------------------------- +// Every give-up after submit is a BilledJobError carrying the cost and the id. +// --------------------------------------------------------------------------- + +test("the deadline throw carries the submit cost and the job id, and keeps its message", async () => { + script = [submit]; + await assert.rejects( + apiKeyAsyncPost("/v1/videos/generations", { prompt: "t" }, { pollBudgetMs: 0, pollIntervalMs: 1 }), + (err: unknown) => { + assert.ok(err instanceof BilledJobError, `expected BilledJobError, got ${String(err)}`); + assert.equal(err.paidUsd, 9.45); + assert.equal(err.jobId, "job_1"); + assert.equal(err.billing, "billed"); + // Unchanged text: isTimeoutError and the wallet-rail message tests key on it. + assert.match(err.message, /Job did not complete within 0s \(last status: queued\)/); + assert.match(err.message, /already been billed to the account; job id job_1/); + assert.match(err.message, /dashboard\/activity before submitting again/); + return true; + }, + ); +}); + +test("an absent cost header leaves paidUsd null (the caller books its estimate), never zero", async () => { + script = [() => submit(null)]; + await assert.rejects( + apiKeyAsyncPost("/v1/audio/generations", { prompt: "t" }, { pollBudgetMs: 0, pollIntervalMs: 1 }), + (err: unknown) => err instanceof BilledJobError && err.paidUsd === null && err.jobId === "job_1", + ); +}); + +test("a non-transient poll status abandons the job WITH the billed note and the cost", async () => { + script = [submit, () => statusOnly(500, {}, { error: "boom" })]; + await assert.rejects( + apiKeyAsyncPost("/v1/videos/generations", { prompt: "t" }, fast), + (err: unknown) => { + assert.ok(err instanceof BilledJobError, `expected BilledJobError, got ${String(err)}`); + assert.match(err.message, /API error 500/); + assert.match(err.message, /"boom"/, "the poll body still reaches the message"); + assert.match(err.message, /already been billed to the account; job id job_1/); + assert.equal(err.paidUsd, 9.45); + assert.equal(err.billing, "billed"); + return true; + }, + ); +}); + +test("a terminal failure the gateway says was NOT charged stays a plain error", async () => { + script = [submit, () => poll("failed", { error: "render exploded", payment_status: "not_charged" })]; + await assert.rejects( + apiKeyAsyncPost("/v1/videos/generations", { prompt: "t" }, fast), + (err: unknown) => { + assert.ok(!(err instanceof BilledJobError), "a refunded failure must not be booked"); + assert.match((err as Error).message, /render exploded.*No payment was taken/); + return true; + }, + ); +}); + +test("a terminal failure with an explicit charged status is billed", async () => { + script = [submit, () => poll("failed", { error: "render exploded", payment_status: "charged" })]; + await assert.rejects( + apiKeyAsyncPost("/v1/videos/generations", { prompt: "t" }, fast), + (err: unknown) => err instanceof BilledJobError && err.billing === "billed" && err.paidUsd === 9.45 && err.jobId === "job_1", + ); +}); + +test("a terminal failure with NO payment status is carried as unknown, still bookable", async () => { + // The gateway contract is to emit payment_status on failures; when it does + // not, we neither claim a refund we did not observe nor drop the cost. + script = [submit, () => poll("failed", { error: "render exploded" })]; + await assert.rejects( + apiKeyAsyncPost("/v1/videos/generations", { prompt: "t" }, fast), + (err: unknown) => { + assert.ok(err instanceof BilledJobError); + assert.equal(err.billing, "unknown"); + assert.equal(err.paidUsd, 9.45); + assert.match(err.message, /Billing status: unknown/); + assert.match(err.message, /for job job_1/); + return true; + }, + ); +}); + +// --------------------------------------------------------------------------- +// The submit itself: no response is not "not billed". +// --------------------------------------------------------------------------- + +test("a submit that aborts says the job MAY have been billed — it does not claim a charge, and does not say retry", async () => { + script = [() => { throw abortError(); }]; + await assert.rejects( + apiKeyAsyncPost("/v1/videos/generations", { prompt: "t" }, fast), + (err: unknown) => { + assert.ok(err instanceof BilledJobError, `expected BilledJobError, got ${String(err)}`); + assert.equal(err.billing, "unknown"); + assert.equal(err.paidUsd, null); + assert.equal(err.jobId, undefined); + assert.match(err.message, /did not return a response/); + assert.match(err.message, /MAY have been accepted and billed/); + assert.match(err.message, /dashboard\/activity before submitting again/); + assert.doesNotMatch(err.message, /already been billed/, "no charge was observed, so none is asserted"); + return true; + }, + ); +}); + +test("a submit that never reached the gateway (DNS / refused) is rethrown as-is — nothing could have been billed", async () => { + for (const code of ["ENOTFOUND", "ECONNREFUSED", "EAI_AGAIN"]) { + script = [() => { const e = new TypeError("fetch failed"); (e as Error & { cause?: unknown }).cause = { code }; throw e; }]; + await assert.rejects( + apiKeyAsyncPost("/v1/videos/generations", { prompt: "t" }, fast), + (err: unknown) => !(err instanceof BilledJobError) && err instanceof TypeError && err.message === "fetch failed", + `cause ${code} must not be reported as a possible charge`, + ); + } +}); diff --git a/test/apps.test.ts b/test/apps.test.ts index a40efde..0cd27a7 100644 --- a/test/apps.test.ts +++ b/test/apps.test.ts @@ -86,3 +86,15 @@ test("the order card knows the tools it calls and the wallet panel its actions", assert.ok(wallet.includes("blockrun_wallet"), "wallet: tool"); for (const action of ["status", "chain", "deposit"]) assert.ok(new RegExp(`["\`']${action}["\`']`).test(wallet), `wallet: action ${action}`); }); + +// The card's two-step confirm used to read the notional from the LAST preview +// while the submitted args read the amount field LIVE — type 50 over a $5 +// quote, skip Re-quote, and "Confirm — sign & submit $5.00" submitted $50. Now +// an edited amount disables Place until re-quoted, and the card renders the +// server's worst-fill bound. The bundle is minified but string literals and +// property names survive verbatim. +test("the order card refuses to place a stale amount and shows the worst fill", () => { + const order = readAppHtml("orderPreview"); + assert.ok(order.includes("Re-quote first"), "order card: stale-amount guard text"); + assert.ok(order.includes("worstFillPrice"), "order card: renders the server's worst-fill bound"); +}); diff --git a/test/budget-limit-env-warning.test.ts b/test/budget-limit-env-warning.test.ts new file mode 100644 index 0000000..1932235 --- /dev/null +++ b/test/budget-limit-env-warning.test.ts @@ -0,0 +1,74 @@ +// Run with: npm test (tsx --test) +// +// BLOCKRUN_BUDGET_LIMIT is sold as the hard stop on every client — including +// the ones with no spend dialog, where it is the ONLY guard. parseBudgetLimitEnv +// maps anything that is not a finite positive number to null, and null means +// UNLIMITED. That contract is shared with BLOCKRUN_CONFIRM_THRESHOLD and stays; +// what must not stay is the silence. An operator who writes "5,00", "5 USD", +// "0" or "-3" gets an unlimited server that looks capped. +// +// The fix is a single stderr line at startup (stderr is the MCP stdio log +// channel; stdout is the protocol). It fires exactly when the env is set to +// something non-empty that parses to null, names the raw value, says the cap +// is OFF, and shows how to write it. It must NOT fire for a valid cap or for +// an unset/blank env — that is the default, not a misconfiguration. +import { test, mock } from "node:test"; +import assert from "node:assert/strict"; +import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { initializeMcpServer } from "../src/mcp-handler.js"; + +// Same fake as apps.test.ts: registration is captured, nothing runs. +function init(env: NodeJS.ProcessEnv): string[] { + const fake = { + registerTool() {}, + registerResource() {}, + } as unknown as McpServer; + const spy = mock.method(console, "error", () => {}); + try { + initializeMcpServer(fake, { argv: ["--profile", "chat"], env }); + return spy.mock.calls.map((c) => c.arguments.map(String).join(" ")); + } finally { + spy.mock.restore(); + } +} + +const budgetLines = (lines: string[]) => lines.filter((l) => l.includes("BLOCKRUN_BUDGET_LIMIT")); + +for (const raw of ["5,00", "5 USD", "5$", "0", "-3", "abc", "NaN", "Infinity"]) { + test(`BLOCKRUN_BUDGET_LIMIT=${JSON.stringify(raw)} warns once that the cap is OFF and names the value`, () => { + const lines = budgetLines(init({ BLOCKRUN_BUDGET_LIMIT: raw })); + assert.equal(lines.length, 1, `expected exactly one warning, got: ${JSON.stringify(lines)}`); + const line = lines[0]; + assert.match(line, /^\[BlockRun\] /, "startup lines carry the [BlockRun] prefix"); + assert.ok(line.includes(`BLOCKRUN_BUDGET_LIMIT="${raw}"`), `the raw value is quoted back: ${line}`); + assert.match(line, /\bOFF\b/, "says the cap is OFF, not 'invalid'"); + assert.match(line, /unlimited/i, "spells out what OFF means for the ledger"); + assert.match(line, /BLOCKRUN_BUDGET_LIMIT=5\b/, "shows a correct spelling to copy"); + assert.match(line, /\$2\.50/, "and that a leading $ is accepted"); + }); +} + +for (const raw of ["5", "5.00", "$2.50", " 10 ", "0.001"]) { + test(`BLOCKRUN_BUDGET_LIMIT=${JSON.stringify(raw)} is a valid cap and stays silent`, () => { + assert.deepEqual(budgetLines(init({ BLOCKRUN_BUDGET_LIMIT: raw })), []); + }); +} + +test("an unset or blank BLOCKRUN_BUDGET_LIMIT is the default, not a misconfiguration — no warning", () => { + assert.deepEqual(budgetLines(init({})), []); + assert.deepEqual(budgetLines(init({ BLOCKRUN_BUDGET_LIMIT: "" })), []); + assert.deepEqual(budgetLines(init({ BLOCKRUN_BUDGET_LIMIT: " " })), []); +}); + +test("the warning does not read process.env when an explicit env is passed", () => { + // initializeMcpServer takes the env it is given; a test (or a host) that + // passes {} must not be judged on the developer's shell. + const saved = process.env.BLOCKRUN_BUDGET_LIMIT; + process.env.BLOCKRUN_BUDGET_LIMIT = "junk"; + try { + assert.deepEqual(budgetLines(init({})), []); + } finally { + if (saved === undefined) delete process.env.BLOCKRUN_BUDGET_LIMIT; + else process.env.BLOCKRUN_BUDGET_LIMIT = saved; + } +}); diff --git a/test/chat-settled-on-error.test.ts b/test/chat-settled-on-error.test.ts index f98bb33..3eb422f 100644 --- a/test/chat-settled-on-error.test.ts +++ b/test/chat-settled-on-error.test.ts @@ -86,6 +86,35 @@ test("a settled-then-failed chat is BOOKED, not silently forgotten", async () => Math.abs(budget.spent - 0.0412) < 1e-9, `settled $0.0412 must be booked; budget.spent = ${budget.spent}`, ); + // ...and the CALLER has to be told. The ledger was right since 0.40.1, but the + // text on this path was a bare "Error: stream failed…" — indistinguishable from + // a free failure, so the obvious next step (retry) settled a second payment. + // The routing loop has said this since 0.40.1; the two direct paths never did. + assert.match(res.content[0].text, /charge stands \(\$0\.041200\)/, res.content[0].text); + assert.match(res.content[0].text, /second charge/i, res.content[0].text); + // The note says "payment". Fed to formatError's keyword classifier that reads + // as an empty wallet — "needs funding" is the exact wrong advice for a call + // that just paid, and it is what the routing loop's note used to earn. + assert.doesNotMatch(res.content[0].text, /needs funding/, res.content[0].text); +}); + +test("a multi-turn chat that settled and then failed says so too", async () => { + script = new Map([["openai/gpt-5.6-terra", { settleUsd: 0.0308, fail: true }]]); + attempts = []; + const { budget, call } = makeHarness(); + + const res = await call({ + message: "and then?", + model: "openai/gpt-5.6-terra", + messages: [{ role: "user", content: "hi" }, { role: "assistant", content: "hello" }], + max_tokens: 1024, + temperature: 1, + }); + + assert.equal(res.isError, true); + assert.match(res.content[0].text, /charge stands \(\$0\.030800\)/, res.content[0].text); + assert.doesNotMatch(res.content[0].text, /needs funding/, res.content[0].text); + assert.ok(Math.abs(budget.spent - 0.0308) < 1e-9, `booked ${budget.spent}`); }); test("a chat that fails BEFORE settling books nothing", async () => { @@ -99,6 +128,8 @@ test("a chat that fails BEFORE settling books nothing", async () => { assert.equal(res.isError, true); assert.equal(budget.spent, 0, `nothing settled, so nothing should be booked (got ${budget.spent})`); + // No money moved, so the text must not say it did — the note is evidence-gated. + assert.doesNotMatch(res.content[0].text, /charge stands/, res.content[0].text); }); test("the routing loop stops after a payment settles — one reservation, one charge", async () => { @@ -121,7 +152,11 @@ test("the routing loop stops after a payment settles — one reservation, one ch "once a payment has settled the loop must stop — retrying charges the caller twice for one tool call", ); assert.equal(res.isError, true); - assert.ok(/charge stands|already been charged|settled/i.test(res.content[0].text), res.content[0].text); + assert.match(res.content[0].text, /charge stands \(\$0\.021700\)/, res.content[0].text); + assert.match(res.content[0].text, /No fallback model was tried/, res.content[0].text); + // Same classifier trap as the direct paths: this note used to end in "your + // wallet needs funding" because formatError saw the word "payment" in it. + assert.doesNotMatch(res.content[0].text, /needs funding/, res.content[0].text); assert.ok(Math.abs(budget.spent - 0.0217) < 1e-9, `booked ${budget.spent}`); }); diff --git a/test/chat-truncation.test.ts b/test/chat-truncation.test.ts index c346e23..2fcfdc8 100644 --- a/test/chat-truncation.test.ts +++ b/test/chat-truncation.test.ts @@ -14,7 +14,7 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { promptCharSize, freeTierTruncationNote } from "../src/tools/chat.js"; -import { FREE_TIER_MAX_PROMPT_CHARS, MODEL_TIERS } from "../src/utils/constants.js"; +import { FREE_TIER_MAX_PROMPT_CHARS, MODEL_TIERS, FREE_CHAT_MODELS } from "../src/utils/constants.js"; const OVER = FREE_TIER_MAX_PROMPT_CHARS + 50_000; const UNDER = FREE_TIER_MAX_PROMPT_CHARS - 1; @@ -32,10 +32,23 @@ test("an oversized prompt on a free model warns, and says how much was lost", () assert.match(note!, /2[0-9]%/); // ~28% discarded }); -test("every model in the free tier is covered by the warning", () => { - for (const m of MODEL_TIERS.free) { +// The cap was measured on the NVIDIA free path (2026-07-21, both alphabets) and +// nowhere else. Since 2026-09-08 free[] also routes two $0 models from other +// vendors (cohere/north-mini-code, poolside/laguna-xs-2.1) whose input handling +// is unmeasured. The warning stays NVIDIA-only on purpose: asserting "a third of +// your prompt was dropped" on a path where it may not have been would push +// agents off a working $0 model onto paid USDC on a false premise — the exact +// harm the byte-vs-character fix below removed. Extend it only with a probe. +test("every NVIDIA model in the free tier is covered by the warning; other vendors are unmeasured and silent", () => { + const nvidia = MODEL_TIERS.free.filter((m) => m.startsWith("nvidia/")); + assert.ok(nvidia.length > 0, "the free tier must still route the measured NVIDIA path"); + for (const m of nvidia) { assert.ok(freeTierTruncationNote(OVER, m), `${m} must be covered`); } + for (const m of MODEL_TIERS.free.filter((m) => !m.startsWith("nvidia/"))) { + assert.ok(FREE_CHAT_MODELS.has(m), `${m} is routed as free, so it must be in FREE_CHAT_MODELS`); + assert.equal(freeTierTruncationNote(OVER, m), null, `${m}: the cap is unmeasured there — do not assert it`); + } }); test("a prompt at or under the cap says nothing", () => { diff --git a/test/chat.test.ts b/test/chat.test.ts index fb68f71..9ce44b1 100644 --- a/test/chat.test.ts +++ b/test/chat.test.ts @@ -3,7 +3,8 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { estimateChatCost, freeTierTruncationNote } from "../src/tools/chat.js"; import { handleAnthropicNative, anthropicCallCost } from "../src/tools/chat-anthropic.js"; -import { MODEL_TIERS, CHAT_PRICE_PER_MTOKEN } from "../src/utils/constants.js"; +import { MODEL_TIERS, CHAT_PRICE_PER_MTOKEN, DEFAULT_CHAT_PRICE, FREE_CHAT_MODELS } from "../src/utils/constants.js"; +import { withTxFee } from "../src/utils/tx-fee.js"; import type { BudgetState } from "../src/types.js"; function newBudget(limit: number | null = null): BudgetState { @@ -93,7 +94,9 @@ test("estimateChatCost keeps the cheap tiers cheaper than the frontier ones", () test("estimateChatCost prices an unknown model at the catalog ceiling, not a guess", () => { // A model added upstream between releases has no table entry. $5/$30 covers - // everything in the catalog except the five pro-tier ids, which ARE listed. + // everything in the catalog except the ids priced above it, which ARE listed + // (seven as of 2026-09-08 — `npm run verify:prices` sweeps the live catalogue + // and fails the moment an eighth appears without a row). const unknown = estimateChatCost(1024, undefined, "someone/brand-new-model", undefined, 100_000); assert.equal(unknown, estimateChatCost(1024, undefined, "openai/gpt-5.6-sol", undefined, 100_000)); assert.ok(unknown >= 0.244171); @@ -155,19 +158,42 @@ test("handleAnthropicNative adds no JSON instruction for plain text", async () = }); // estimateChatCost reserves $0 for mode:"free" with no model to override it. That -// is only sound because every free[] entry is an nvidia/* model the gateway serves -// at $0 — an unenforced invariant on a hand-edited array that has now been -// rewritten in three consecutive releases (0.31.x, 0.32.0, 0.32.1). One paid model -// landing in free[] silently switches the budget gate off for mode:"free", and -// every other test here still passes. Pin it. -test("every MODEL_TIERS.free entry is an nvidia/* model (keeps the $0 reserve honest)", () => { +// is only sound because every free[] entry is a model the gateway serves at $0 — +// an unenforced invariant on a hand-edited array that has now been rewritten in +// four releases (0.31.x, 0.32.0, 0.32.1, and this one). One paid model landing +// in free[] silently switches the budget gate off for mode:"free", and every +// other test here still passes. Pin it. +// +// This used to pin "every free[] entry is nvidia/*", and the $0 classifier was +// the same vendor test. Both are wrong-by-design since the 2026-09-08 catalogue: +// cohere/north-mini-code and poolside/laguna-xs-2.1 are billed $0 on both +// gateways, so an explicit call to either reserved the $5/$30 default and an +// exhausted budget refused a free call. FREE_CHAT_MODELS is the set now; the +// catalogue sweep in `npm run verify:prices` fails if any member starts costing. +test("every MODEL_TIERS.free entry is in FREE_CHAT_MODELS, and every member reserves $0", () => { assert.ok(MODEL_TIERS.free.length > 0, "free tier must not be empty"); for (const m of MODEL_TIERS.free) { - assert.ok( - m.startsWith("nvidia/"), - `${m} is in the free tier but is not nvidia/* — estimateChatCost would reserve $0 for a paid model`, - ); + assert.ok(FREE_CHAT_MODELS.has(m), `${m} is routed as free but FREE_CHAT_MODELS does not list it — estimateChatCost would reserve for a $0 call`); + } + for (const m of FREE_CHAT_MODELS) { + assert.equal(estimateChatCost(1024, undefined, m, undefined, 600 * 1024), 0, m); + // The bare spelling is a real, chargeable-or-free id too (see BARE_TO_PREFIXED). + assert.equal(estimateChatCost(1024, undefined, m.slice(m.indexOf("/") + 1), undefined, 600 * 1024), 0, `bare ${m}`); + } +}); + +test("a $0 model outside nvidia/ is free — the classifier is membership, not vendor", () => { + // Live billing_mode:"free" on both gateways, 2026-09-08. Before FREE_CHAT_MODELS + // both reserved the unknown-model ceiling (the gpt-5.6-sol figure). + for (const id of ["cohere/north-mini-code", "poolside/laguna-xs-2.1", "north-mini-code", "laguna-xs-2.1"]) { + assert.equal(estimateChatCost(1024, undefined, id, undefined, 100_000), 0, id); } + // ...and a paid model from the same vendors is not swept along. + assert.ok(estimateChatCost(1024, undefined, "cohere/command-a", undefined, 100_000) > 0); + // The truncation warning stays NVIDIA-only: the 128 KiB silent cap was measured + // on that path and nowhere else, so it is not asserted for cohere/poolside. + assert.equal(freeTierTruncationNote(200_000, "cohere/north-mini-code"), null); + assert.ok(freeTierTruncationNote(200_000, "nvidia/nemotron-3-ultra-550b")); }); // A tier that empties out resolves MODEL_TIERS[mode][0] to undefined, which sends @@ -244,7 +270,7 @@ test("anthropicCallCost honours the $0.001 floor and the prefixed/bare id", () = anthropicCallCost("claude-opus-5", 10_000, 1024), anthropicCallCost("anthropic/claude-opus-5", 10_000, 1024), ); - // A date-suffixed id still resolves via the prefix match. + // A date-suffixed id still resolves — the suffix is stripped, not prefix-matched. assert.ok(anthropicCallCost("claude-opus-5-20260101", 2, 1024) !== null); // An unknown model returns null so the caller falls back to the estimate, // rather than inventing a number. @@ -307,7 +333,7 @@ test("a vendor-less free model is still free, and still warns about truncation", }); test("every catalog id has a unique vendor-less segment — the mapping cannot be ambiguous", () => { - const ids = [...Object.keys(CHAT_PRICE_PER_MTOKEN), ...Object.values(MODEL_TIERS).flat()].filter((i) => i.includes("/")); + const ids = [...Object.keys(CHAT_PRICE_PER_MTOKEN), ...Object.values(MODEL_TIERS).flat(), ...FREE_CHAT_MODELS].filter((i) => i.includes("/")); const byBare = new Map>(); for (const id of ids) { const bare = id.slice(id.indexOf("/") + 1); @@ -317,3 +343,115 @@ test("every catalog id has a unique vendor-less segment — the mapping cannot b const clashes = [...byBare.entries()].filter(([, set]) => set.size > 1); assert.deepEqual(clashes, [], `two vendors ship the same model name: ${JSON.stringify(clashes)}`); }); + +// ── Two flagship ids landed above the default, and nothing noticed ── +// +// Live GET /v1/models on BOTH gateways, 2026-09-08: openai/gpt-6-astra and +// anthropic/claude-fable-5.1 are $10/$50, available, and had no table row, so an +// explicit `model` fell to DEFAULT_CHAT_PRICE ($5/$30) — 2x short on input, +// 1.67x on output, and the confirm-spend prompt showed the same wrong figure. +// This is the exact defect class the table was introduced to close; the table +// header even asserted it held "every catalog model priced ABOVE the default". +// `npm run verify:prices` now sweeps the live catalogue for this shape and fails +// on any above-default id without a row; these pin the two rows that closed it. +test("gpt-6-astra and claude-fable-5.1 reserve at their live $10/$50, not the $5/$30 default", () => { + assert.deepEqual(CHAT_PRICE_PER_MTOKEN["openai/gpt-6-astra"], { input: 10, output: 50 }); + assert.deepEqual(CHAT_PRICE_PER_MTOKEN["anthropic/claude-fable-5.1"], { input: 10, output: 50 }); + + const fable5 = estimateChatCost(1024, undefined, "anthropic/claude-fable-5", undefined, 100_000); + const unknown = estimateChatCost(1024, undefined, "someone/brand-new-model", undefined, 100_000); + // Bare spellings ride along: BARE_TO_PREFIXED is derived from the table keys. + for (const id of ["openai/gpt-6-astra", "anthropic/claude-fable-5.1", "gpt-6-astra", "claude-fable-5.1"]) { + const reserved = estimateChatCost(1024, undefined, id, undefined, 100_000); + // 100k chars at 2 chars/token = 50k input tokens at $10/M; 1024 output at $50/M. + assert.equal(reserved, withTxFee((50_000 / 1e6) * 10 + (1024 / 1e6) * 50), id); + assert.equal(reserved, fable5, `${id} shares fable-5's $10/$50 rate`); + assert.ok(reserved > unknown, `${id} must reserve above the unknown-model default (${reserved} vs ${unknown})`); + // fable-5's live settle for this exact call (LIVE_CHARGE_100K above): same + // rate, same gateway formula, so the same charge has to be covered. + assert.ok(reserved >= 0.486310, id); + } + // And the default itself did not move: raising it would double the reserve for + // every genuinely unknown non-pro model and lock small budgets out. + assert.deepEqual(DEFAULT_CHAT_PRICE, { input: 5, output: 30 }); +}); + +// ── thinking.budget_tokens is only ever SENT on the native claude-* path ── +// +// The schema says "Ignored for non-Claude models", and the handler honours that: +// only handleAnthropicNative receives `thinking`; the multi-turn, explicit-model +// and routing paths build their options from {maxTokens, temperature, +// responseFormat, stop}. But the reserve folded budget_tokens into output +// unconditionally, so mode:"powerful" + a 100k budget reserved ~$18 (gpt-5.4-pro +// at $180/M) for a call that settles at cents — a spurious refusal for a +// delegated agent, and a wrong "Estimated: $X" shown to a human under +// BLOCKRUN_CONFIRM_SPEND. Over-reserve is the safe direction for the gate, but +// a number a human is asked to approve has to be the number the call can settle at. +test("estimateChatCost ignores thinking.budget_tokens for non-Claude models, as the schema promises", () => { + assert.equal(estimateChatCost(1024, "cheap", undefined, 100_000), estimateChatCost(1024, "cheap", undefined)); + assert.equal(estimateChatCost(1024, "powerful", undefined, 100_000), estimateChatCost(1024, "powerful", undefined)); + assert.equal(estimateChatCost(1024, undefined, "openai/gpt-5.6-terra", 100_000), estimateChatCost(1024, undefined, "openai/gpt-5.6-terra")); + assert.equal(estimateChatCost(1024, undefined, "gpt-5.4-pro", 100_000), estimateChatCost(1024, undefined, "gpt-5.4-pro")); + // Every Claude spelling still folds — canonicalChatModel + isAnthropicModel + // agree with the dispatch in the handler, so prefixed and bare both count. + assert.ok(estimateChatCost(1024, undefined, "claude-opus-4.8", 100_000) > estimateChatCost(1024, undefined, "claude-opus-4.8") * 10); + assert.ok(estimateChatCost(1024, undefined, "anthropic/claude-fable-5.1", 100_000) > estimateChatCost(1024, undefined, "anthropic/claude-fable-5.1") * 10); +}); + +// ── The native ledger keys on the CATALOGUE spelling; the gateway echoes another ── +// +// /v1/messages echoes the UPSTREAM id (blockrun's ANTHROPIC_MODEL_MAP), which is +// dashed and often dated: claude-fable-5-1, claude-haiku-4-5-20251001, +// claude-sonnet-4-5-20250929. The table is dotted: anthropic/claude-fable-5.1. +// The old lookup fell back to a startsWith prefix match, which (a) never fired +// for those dashed echoes — every one silently booked the pre-call estimate +// instead of the reconstructed quote — and (b) DID fire on a version suffix, so +// "claude-fable-5-1" booked claude-fable-5's row: right by coincidence today +// (both $10/$50), and a sibling priced differently from its major would book +// the wrong number with no signal, because "null -> estimate" never engages +// when a rate WAS found. On this path the table IS the ledger. +test("anthropicCallCost normalises the gateway's echo to the catalogue key and never prefix-matches", () => { + const ECHOES: Array<[string, string]> = [ + ["claude-fable-5-1", "anthropic/claude-fable-5.1"], + ["claude-fable-5.1", "anthropic/claude-fable-5.1"], + ["claude-haiku-4-5-20251001", "anthropic/claude-haiku-4.5"], + ["claude-opus-4-8", "anthropic/claude-opus-4.8"], + ["claude-sonnet-4-5-20250929", "anthropic/claude-sonnet-4.5"], + ["claude-sonnet-4.6-20260301", "anthropic/claude-sonnet-4.6"], + ["claude-opus-5-20260101", "anthropic/claude-opus-5"], + ["anthropic/claude-opus-5", "anthropic/claude-opus-5"], + ]; + for (const [echo, key] of ECHOES) { + const booked = anthropicCallCost(echo, 100_000, 1024); + assert.ok(booked !== null, `${echo} must resolve to a row`); + assert.equal(booked, anthropicCallCost(key, 100_000, 1024), `${echo} -> ${key}`); + } + // Rows stay distinct: haiku's echo books haiku, not a dearer sibling. + assert.ok(anthropicCallCost("claude-haiku-4-5-20251001", 100_000, 1024)! < anthropicCallCost("claude-sonnet-4-5-20250929", 100_000, 1024)!); + // A sibling with NO row returns null — the estimate fallback — rather than + // borrowing its major version's rate. This is the mechanism that would have + // let claude-fable-5.1 book fable-5's row before it had one of its own. + for (const unlisted of ["claude-sonnet-5-1", "claude-sonnet-5.1", "claude-fable-5-2", "claude-opus-5-5-20270101", "claude-opus-5x", "claude-does-not-exist"]) { + assert.equal(anthropicCallCost(unlisted, 100_000, 1024), null, `${unlisted} has no row and must not borrow one`); + } +}); + +// ── The Anthropic rows are the native LEDGER, so a stale-high row is not "safe" ── +// +// On the OpenAI-compat paths a row above the live rate only over-reserves (the +// gate is tighter than it needs to be; recordActualSpend books the real settle). +// On the native /v1/messages path there is no settlement counter to read, so +// anthropicCallCost books THIS TABLE — and both gateways cut claude-sonnet-5 to +// $2/$10 while the row stayed at $3/$15, a 1.5x over-count on every call that +// tripped budget caps at two-thirds of their real allowance. Live GET /v1/models, +// both gateways, 2026-09-08. The catalogue sweep now fails on an anthropic/* row +// above the Base rate for exactly this reason. +test("claude-sonnet-5 books at the gateway's $2/$10, not the old $3/$15", () => { + assert.deepEqual(CHAT_PRICE_PER_MTOKEN["anthropic/claude-sonnet-5"], { input: 2, output: 10 }); + // 100k chars -> ceil(100000/2.08)+20 = 48,097 input tokens at $2/M = $0.096194; + // 1024 max_tokens x 0.1 = 102.4 output tokens at $10/M = $0.001024; + // + the $0.001 observed fee, ceiled to a micro-USDC = $0.098218. + assert.equal(anthropicCallCost("claude-sonnet-5", 100_000, 1024), 0.098218); + // The old row booked $0.146827 for the same call. + assert.ok(anthropicCallCost("claude-sonnet-5", 100_000, 1024)! < 0.146827 * 0.7); +}); diff --git a/test/confirm-spend-coverage.test.ts b/test/confirm-spend-coverage.test.ts index 09bde45..485f909 100644 --- a/test/confirm-spend-coverage.test.ts +++ b/test/confirm-spend-coverage.test.ts @@ -9,9 +9,13 @@ // // Two guards, because the failure mode is silence in both directions: // -// 1. STATIC — every src/tools/*.ts that reserves budget must also call -// confirmSpend. A new paid tool that copies the reserve/record shape but -// forgets the confirm would otherwise ship un-gated forever. +// 1. STATIC — every src/tools/*.ts that can PAY must reserve budget AND call +// confirmSpend. "Can pay" is read off the imports: a paid SDK client from +// utils/wallet.ts, or one of the hand-built rails (raw-call, api-key-call, +// solana-402). The guard used to key on reserveBudget alone and skip any +// file without it — so the worst possible offender, a tool that pays and +// never reserves, was the one it could not see. A file that reserves via +// some helper this list does not know is still held to the confirm. // 2. BEHAVIORAL — with confirmation on and a client that answers "decline", // every paid tool must return a non-error "declined" result, release its // reservation (budget.spent back to 0), and never reach the network. @@ -34,23 +38,89 @@ const TOOLS_DIR = join(ROOT, "src", "tools"); // --------------------------------------------------------------------------- // 1. Static guard // --------------------------------------------------------------------------- -test("every tool that reserves budget also asks the user (confirmSpend)", () => { +// The surfaces through which a tool file can move money. A file that imports +// any of these is a paid tool, whether or not it remembered to reserve. +const PAID_CLIENTS = /\b(getClient|getImageClient|buildClient|buildClientWithTimeout|getAnthropicClient|getPriceClient)\b/; +const PAID_HELPERS = /from "\.\.\/utils\/(raw-call|api-key-call|solana-402)\.js"/; + +// Paid-surface importers that genuinely never charge. Each entry is a claim +// that has to be re-made when the file changes; keep it short and say why. +const FREE_BY_DESIGN: Record = { + // getClient() only feeds loadModels()/listModels — the free catalogue GET. + "models.ts": "getClient feeds the free /v1/models catalogue read only", +}; + +/** + * What the static guard sees in one tool file. Exported shape, so the guard's + * own rules can be tested on fixtures below — a guard that cannot be shown to + * bite is only a comment. + */ +function classifyToolSource(file: string, src: string): { paid: boolean; reserves: number; confirms: number; imports: boolean; offence: string | null } { + const walletImport = /import\s*\{([^}]*)\}\s*from\s*"\.\.\/utils\/wallet\.js"/.exec(src)?.[1] ?? ""; + const paidSurface = PAID_CLIENTS.test(walletImport) || PAID_HELPERS.test(src); + const reserves = (src.match(/reserveBudget\(budget/g) ?? []).length; + const confirms = (src.match(/confirmSpend\(server/g) ?? []).length; + const imports = /from "\.\.\/utils\/confirm-spend\.js"/.test(src); + const paid = paidSurface || reserves > 0; + if (!paid) return { paid, reserves, confirms, imports, offence: null }; + if (paidSurface && reserves === 0 && file in FREE_BY_DESIGN) return { paid, reserves, confirms, imports, offence: null }; + // No parity requirement between reserves and confirms: speech, video and + // image legitimately RE-reserve inside a 402 onQuote after the one confirm. + let offence: string | null = null; + if (reserves === 0) offence = "pays but never reserves budget"; + else if (!imports || confirms === 0) offence = "reserves budget but never asks (confirmSpend)"; + return { paid, reserves, confirms, imports, offence }; +} + +test("every tool that can pay reserves budget AND asks the user (confirmSpend)", () => { const offenders: string[] = []; for (const file of readdirSync(TOOLS_DIR).filter((f) => f.endsWith(".ts"))) { - const src = readFileSync(join(TOOLS_DIR, file), "utf8"); - const reserves = (src.match(/reserveBudget\(budget/g) ?? []).length; - if (reserves === 0) continue; - const imports = /from "\.\.\/utils\/confirm-spend\.js"/.test(src); - const calls = (src.match(/confirmSpend\(server/g) ?? []).length; - if (!imports || calls === 0) offenders.push(`${file} (reserves=${reserves}, confirms=${calls})`); + const c = classifyToolSource(file, readFileSync(join(TOOLS_DIR, file), "utf8")); + if (c.offence) offenders.push(`${file}: ${c.offence} (reserves=${c.reserves}, confirms=${c.confirms})`); } assert.deepEqual( offenders, [], - `paid tools that charge without confirmSpend — they bypass BLOCKRUN_CONFIRM_SPEND:\n ${offenders.join("\n ")}`, + `paid tools that bypass the budget cap or BLOCKRUN_CONFIRM_SPEND:\n ${offenders.join("\n ")}`, ); }); +test("the FREE_BY_DESIGN allowlist names only files that still exist and still import a paid surface", () => { + // A stale entry is a hole: rename models.ts, add a paid call to the new + // file, and the old name would keep excusing nothing while the new one is + // judged normally — fine. But an entry whose file no longer imports a paid + // surface is dead weight that invites copy-paste, so it must go. + for (const file of Object.keys(FREE_BY_DESIGN)) { + const src = readFileSync(join(TOOLS_DIR, file), "utf8"); + const walletImport = /import\s*\{([^}]*)\}\s*from\s*"\.\.\/utils\/wallet\.js"/.exec(src)?.[1] ?? ""; + assert.ok(PAID_CLIENTS.test(walletImport) || PAID_HELPERS.test(src), `${file} no longer imports a paid surface — drop it from FREE_BY_DESIGN`); + assert.equal((src.match(/reserveBudget\(budget/g) ?? []).length, 0, `${file} now reserves budget — it is a paid tool, drop it from FREE_BY_DESIGN`); + } +}); + +test("the static guard bites: a tool that pays without reserving, or reserves without asking, is an offender", () => { + const RESERVE = "const gate = reserveBudget(budget, agent_id, 0.01);"; + const CONFIRM = 'import { confirmSpend } from "../utils/confirm-spend.js";\nconst c = await confirmSpend(server, { usd: 0.01, label: "x" });'; + const client = 'import { getClient } from "../utils/wallet.js";'; + const helper = 'import { apiKeyPost } from "../utils/api-key-call.js";'; + const freeWallet = 'import { getWalletInfo, getChain } from "../utils/wallet.js";'; + + // The hole this test closes: pays via a client, never reserves → was skipped. + assert.equal(classifyToolSource("new.ts", `${client}\n${CONFIRM}`).offence, "pays but never reserves budget"); + assert.equal(classifyToolSource("new.ts", `${helper}`).offence, "pays but never reserves budget"); + // The original rule, still enforced. + assert.equal(classifyToolSource("new.ts", `${client}\n${RESERVE}`).offence, "reserves budget but never asks (confirmSpend)"); + assert.equal(classifyToolSource("new.ts", `${RESERVE}`).offence, "reserves budget but never asks (confirmSpend)", "reserving via an unknown helper is still held to the confirm"); + // Compliant, including the legitimate re-reserve pattern (2 reserves, 1 confirm). + assert.equal(classifyToolSource("new.ts", `${client}\n${RESERVE}\n${CONFIRM}`).offence, null); + assert.equal(classifyToolSource("new.ts", `${helper}\n${RESERVE}\n${RESERVE}\n${CONFIRM}`).offence, null); + // Free tools: wallet-status imports and no rail are not paid at all. + assert.deepEqual(classifyToolSource("free.ts", `${freeWallet}`), { paid: false, reserves: 0, confirms: 0, imports: false, offence: null }); + // The allowlist excuses a paid-surface importer only under its own name. + assert.equal(classifyToolSource("models.ts", `${client}`).offence, null); + assert.equal(classifyToolSource("models-v2.ts", `${client}`).offence, "pays but never reserves budget"); +}); + // --------------------------------------------------------------------------- // 2. Behavioral guard // --------------------------------------------------------------------------- @@ -71,6 +141,11 @@ mock.module("../src/utils/wallet.js", { buildClientWithTimeout: () => trap, getPriceClient: () => trap, getAnthropicClient: () => trap, + // blockrun_image: its Base rail is the SDK ImageClient, and image.ts + // statically imports utils/solana-402.ts, which resolves the Solana key + // through wallet.ts (image-cost.test.ts documents the same two exports). + getImageClient: () => trap, + resolveSolanaKey: () => undefined, baseOnlyMessage: () => null, getOrCreateWalletKey: () => TEST_KEY, getWalletInfo: async () => ({ address: "0xTEST" }), @@ -107,12 +182,31 @@ const CASES: Array<{ tool: string; mod: string; register: string; args: Record { throw new Error("UNEXPECTED_WALLET_USE"); }; +const trap = new Proxy({}, { get: () => boom }); +mock.module("../src/utils/wallet.js", { + namedExports: { + getApiBase: () => "https://blockrun.ai/api", + resolveGatewayUrl: (u: string) => u, + getChain: () => "base", + getClient: () => trap, + buildClient: () => trap, + buildClientWithTimeout: () => trap, + getPriceClient: () => trap, + getAnthropicClient: () => trap, + baseOnlyMessage: () => null, + getOrCreateWalletKey: () => boom(), + getWalletInfo: async () => boom(), + }, +}); + +const requests: string[] = []; +mock.module("../src/utils/http.js", { + namedExports: { + fetchWithTimeout: async (url: string) => { + requests.push(url); + return { ok: true, status: 200, json: async () => ({ pairs: [] }) }; + }, + isTimeoutError: () => false, + }, +}); + +type Handler = (args: Record) => Promise<{ content: Array<{ type: string; text?: string }>; isError?: boolean }>; + +async function harness() { + const { registerDexTool, parseTokenAddresses } = await import("../src/tools/dex.js"); + let handler: Handler | undefined; + let name = ""; + registerDexTool({ registerTool: (n: string, _c: unknown, h: Handler) => { name = n; handler = h; } } as never); + assert.ok(handler, "blockrun_dex did not register"); + requests.length = 0; + return { name, handler: handler!, parseTokenAddresses }; +} + +const EVM = "0x6982508145454Ce325dDbE47a25d4ec3d2311933"; // PEPE +const SOL = "So11111111111111111111111111111111111111112"; // wSOL + +test("blockrun_dex stays registered under its name", async () => { + const { name } = await harness(); + assert.equal(name, "blockrun_dex"); +}); + +test("a well-formed EVM or Solana address reaches DexScreener unchanged", async () => { + const { handler } = await harness(); + await handler({ token: EVM }); + await handler({ token: SOL }); + assert.deepEqual(requests, [ + `https://api.dexscreener.com/latest/dex/tokens/${EVM}`, + `https://api.dexscreener.com/latest/dex/tokens/${SOL}`, + ]); +}); + +test("whitespace is trimmed and a comma-separated list is passed through (DexScreener takes up to 30)", async () => { + const { handler } = await harness(); + await handler({ token: ` ${EVM} , ${SOL} ` }); + assert.deepEqual(requests, [`https://api.dexscreener.com/latest/dex/tokens/${EVM},${SOL}`]); +}); + +test("a Sui coin type (with `::`) is accepted — the shape is an allow-list, not an EVM/Solana whitelist", async () => { + const { handler } = await harness(); + await handler({ token: "0x2::sui::SUI" }); + assert.equal(requests.length, 1); + assert.ok(requests[0].endsWith("/tokens/0x2::sui::SUI")); +}); + +test("anything URL syntax could reinterpret is refused before any request, as an error result", async () => { + const { handler } = await harness(); + const bad = [ + "../search?q=pepe", // path traversal + query + "abc#x", // fragment drops the tail + "a/b", // extra path segment + "..", // parent segment + ".", // current segment + ".hidden", // dot-led segment + "pepe?chain=solana", // query + "0x1234%2F..", // percent-encoding + "a b", // whitespace inside + "So11+111", // plus + "x", // one char + " ", // blank + ",,,", // only separators + Array.from({ length: 31 }, () => EVM).join(","), // over the 30-address limit + ]; + for (const token of bad) { + const res = await handler({ token }); + assert.equal(res.isError, true, `${JSON.stringify(token)}: must be an error`); + assert.match(res.content[0].text ?? "", /Invalid token address/, `${JSON.stringify(token)}: names the problem`); + assert.match(res.content[0].text ?? "", /use query instead/, `${JSON.stringify(token)}: points at the search branch`); + } + assert.deepEqual(requests, [], "no request was sent for any rejected token"); +}); + +test("an empty token is not an address at all — it falls through to the existing 'provide something' error", async () => { + const { handler } = await harness(); + const res = await handler({ token: "" }); + assert.equal(res.isError, true); + assert.match(res.content[0].text ?? "", /Provide query, token address, or symbol/); + assert.deepEqual(requests, []); +}); + +test("parseTokenAddresses: the shape guard in isolation", async () => { + const { parseTokenAddresses } = await harness(); + assert.deepEqual(parseTokenAddresses(EVM), [EVM]); + assert.deepEqual(parseTokenAddresses(`${EVM},${SOL}`), [EVM, SOL]); + assert.deepEqual(parseTokenAddresses("EQCxE6mUtQJKFnGfaROTKOt1lZbDiiX1kCixRv7Nw2Id_sDs"), ["EQCxE6mUtQJKFnGfaROTKOt1lZbDiiX1kCixRv7Nw2Id_sDs"]); // TON, with _ and - + assert.equal(parseTokenAddresses("../x"), null); + assert.equal(parseTokenAddresses("a".repeat(201)), null, "200-char cap per address"); + assert.deepEqual(parseTokenAddresses("a".repeat(200)), ["a".repeat(200)]); +}); + +test("the query/symbol branch is unchanged and still encodes", async () => { + const { handler } = await harness(); + await handler({ query: "pepe coin&x" }); + assert.deepEqual(requests, ["https://api.dexscreener.com/latest/dex/search?q=pepe%20coin%26x"]); +}); diff --git a/test/errors.test.ts b/test/errors.test.ts index 2be8603..c0a6abb 100644 --- a/test/errors.test.ts +++ b/test/errors.test.ts @@ -1,7 +1,7 @@ // Run with: npm test (tsx --test) import { test } from "node:test"; import assert from "node:assert/strict"; -import { formatError, isPaymentRejectionError } from "../src/utils/errors.js"; +import { extractErrorMessage, formatError, isPaymentRejectionError } from "../src/utils/errors.js"; test("model-unavailable (token360) → steers to a sibling model, not a generic blip", () => { const msg = "Video generation failed: API error 500: token360 video submit failed: Model 'seedance-2.0-fast' not found or not active for requested provider"; @@ -114,3 +114,87 @@ test("isPaymentRejectionError matches settlement failures, not outage status tex assert.equal(isPaymentRejectionError('Unexpected response 500 (expected a 402 payment challenge): {"error":"bad gateway"}'), false); assert.equal(isPaymentRejectionError("Unexpected response 425 (expected a 402 payment challenge): liveness not finished"), false); }); + +// --- blockrun-mcp#132: the gateway said "payment NOT charged"; the tool said "after payment" --- + +class FakeAPIError extends Error { + constructor(message: string, public statusCode: number, public response: unknown) { + super(message); + } +} + +test("extractErrorMessage surfaces the SDK's `detail` field (blockrun-llm-ts#39)", () => { + // Post-#39 sanitizer output: `message` is the gateway's top-level `error`, + // `detail` is the gateway's own `message` — the cause + settlement status. + const err = new FakeAPIError("API error after payment: 502", 502, { + message: "Upstream provider error", + detail: "Predexon 500: An unexpected error occurred (payment NOT charged)", + }); + const msg = extractErrorMessage(err); + assert.match(msg, /Upstream provider error/); + assert.match(msg, /payment NOT charged/); + // …and formatError says so in its OWN words. Only the guidance after the echoed + // message can prove that: the input already contains "payment NOT charged", so + // a whole-output /not charged/ match — or a doesNotMatch(/needs funding/) on a + // 5xx, whose branch is tested before the funding one — can never fail. This + // is the generic path every tool without a bespoke formatter goes through. + const out = formatError(msg); + const guidance = out.slice(out.indexOf("\n\n")); + assert.match(guidance, /The gateway reported that this call was not settled — nothing was charged\./); + // The outage advice stays — an uncharged upstream 5xx is reasonable to retry. + assert.match(guidance, /temporary API issue/); + assert.doesNotMatch(guidance, /needs funding/); +}); + +test("a post-payment 5xx WITHOUT the gateway's uncharged marker never claims nothing was charged", () => { + // The formatter must never invent a settlement claim: only the gateway's own + // marker in the message earns the "nothing was charged" line. + for (const msg of [ + "API error after payment: 502\nRequest failed", + "API error after payment: 500 Internal Server Error", + "error 500 occurred", + ]) { + const out = formatError(msg); + assert.match(out, /temporary API issue/, msg); + assert.doesNotMatch(out, /nothing was charged|not settled/, msg); + } +}); + +test("extractErrorMessage does not repeat a detail identical to the message", () => { + const err = new FakeAPIError("API error: 400", 400, { message: "Bad request", detail: "Bad request" }); + assert.equal(extractErrorMessage(err).match(/Bad request/g)?.length, 1); +}); + +test("extractErrorMessage is unchanged for the pre-#39 shape (message + code only)", () => { + const err = new FakeAPIError("API error after payment: 502", 502, { message: "Request failed", code: "x" }); + assert.equal(extractErrorMessage(err), "API error after payment: 502\nRequest failed"); +}); + +test("a 501 is 'not served', not a temporary outage to retry", () => { + // Live 2026-09-08: GET /v1/stocks/us/price/AAPL → 501 before any 402. + const out = formatError("API error: 501\nUS Stock price is not available"); + assert.match(out, /does not serve this endpoint/); + assert.match(out, /nothing was charged/); + assert.doesNotMatch(out, /temporary API issue/); + assert.doesNotMatch(out, /needs funding/); +}); + +test("a 501-shaped number inside a message is not read as a status", () => { + const out = formatError("API error 500: batch of 501 items rejected"); + assert.match(out, /temporary API issue/); + assert.doesNotMatch(out, /does not serve this endpoint/); +}); + +test("a post-payment 501 does not claim nothing was charged", () => { + const out = formatError("API error after payment: 501\nRequest failed"); + assert.match(out, /does not serve this endpoint/); + assert.doesNotMatch(out, /nothing was charged/); + assert.match(out, /whether this call settled/); + assert.doesNotMatch(out, /temporary API issue/); +}); + +test("the SDK's post-payment prefix counts as a labelled status", () => { + // "API error after payment: 502" — the word before the number is "payment". + const out = formatError("API error after payment: 502\nRequest failed"); + assert.match(out, /temporary API issue/); +}); diff --git a/test/image-cost.test.ts b/test/image-cost.test.ts index bfcc657..f1283bd 100644 --- a/test/image-cost.test.ts +++ b/test/image-cost.test.ts @@ -1,11 +1,35 @@ -// Run with: npm test (tsx --test) +// Run with: npm test (tsx --experimental-test-module-mocks --test) // Verifies the Cost footer added to blockrun_image, without any real spend: -// the paid ImageClient and the chain selector are mocked, then the registered -// handler is invoked and its text/structured output is asserted. +// the auth rail is pinned to wallet/Base, the paid ImageClient and the chain +// selector are mocked, and the shared fetch helper is a trap — then the +// registered handler is invoked and its text/structured output is asserted. import { test, mock } from "node:test"; import assert from "node:assert/strict"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; import type { BudgetState } from "../src/types.js"; +// Pin the RAIL before anything can load utils/auth.ts. image.ts asks +// isApiKeyMode() BEFORE it consults the mocked getChain()/getImageClient(), and +// that answer comes from the developer's own BLOCKRUN_API_KEY / ~/.blockrun/.api-key +// — so on a machine set up for account mode the six handler calls below used to +// leave the mocks entirely and POST to the gateway with the real Bearer key. +// Same discipline as auth-mode.test.ts: a temp HOME (auth.ts captures the key +// file path from os.homedir() at import time) and no env key. auth.js itself is +// NOT mocked — onramp.ts (imported by image.ts) needs PORTAL_CREDITS_URL from +// it, and a partial namedExports mock fails to link. +const home = fs.mkdtempSync(path.join(os.tmpdir(), "blockrun-image-cost-")); +const realHome = process.env.HOME; +const savedApiKey = process.env.BLOCKRUN_API_KEY; +process.env.HOME = home; +delete process.env.BLOCKRUN_API_KEY; +process.on("exit", () => { + if (realHome === undefined) delete process.env.HOME; else process.env.HOME = realHome; + if (savedApiKey === undefined) delete process.env.BLOCKRUN_API_KEY; else process.env.BLOCKRUN_API_KEY = savedApiKey; + fs.rmSync(home, { recursive: true, force: true }); +}); + // Mock the wallet module BEFORE importing the tool: force Base chain and hand // back a fake ImageClient whose generate/edit resolve to a hosted URL (no // network, no payment). @@ -27,8 +51,26 @@ mock.module("../src/utils/wallet.js", { resolveSolanaKey: () => undefined, }, }); +// Belt and braces: every rail that is not the mocked ImageClient (account +// apiKeyPost, Solana manual x402) bottoms out in this helper. If a future +// change routes past the pin above, the test fails HERE, for the right reason, +// instead of reaching the network. +let networkCalls = 0; +mock.module("../src/utils/http.js", { + namedExports: { + fetchWithTimeout: async () => { networkCalls++; throw new Error("network call escaped the mocks"); }, + isTimeoutError: () => false, + }, +}); const { registerImageTool, estimateCost } = await import("../src/tools/image.js"); +const { isApiKeyMode } = await import("../src/utils/auth.js"); + +test("the suite runs on the wallet rail whatever the developer's account setup", () => { + // If this fails, every handler test below is exercising the account rail — + // and without the pin, a real key. + assert.equal(isApiKeyMode(), false); +}); // Minimal McpServer stub: capture the handler registerImageTool installs. function makeHarness() { @@ -53,6 +95,7 @@ test("generate result includes a Cost line at the CHARGED price, not the catalog assert.match(text, /Cost: \$0\.0650/); // 0.06 catalog x 1.05 + $0.002 assert.equal(res.structuredContent.cost_usd, 0.065); assert.equal(res.isError, undefined); + assert.equal(networkCalls, 0, "the mocked ImageClient must be the only rail this suite touches"); }); test("large gpt-image-2 render is billed at the large-size CHARGED price", async () => { diff --git a/test/image-edit-label.test.ts b/test/image-edit-label.test.ts new file mode 100644 index 0000000..97f27fa --- /dev/null +++ b/test/image-edit-label.test.ts @@ -0,0 +1,194 @@ +// Run with: npm test (tsx --experimental-test-module-mocks --test) +// +// blockrun_image edit reads whatever local file the model names — that is the +// documented feature ("edit ~/Downloads/photo.png") — base64s it into the +// request body and ships it to the gateway. The confirmSpend dialog is the one +// moment a human sees the call before it leaves the machine, and its label used +// to say only `image edit · `: which file was about to leave was not on +// it. A prompt-injected reference to ~/Pictures/IMG_1234.jpg looked exactly like +// a normal $0.05 edit. +// +// Two properties are pinned here: +// - the local branch resolves through fs.realpath, so the label carries the +// REAL path — a symlink named innocently is shown as what it points at; +// - the label lists every local source and the mask, and says nothing about +// files for data: URIs or a plain generate. +// Nothing is restricted: cwd, tmpdir, home all still work. This is disclosure, +// not a sandbox. +process.env.BLOCKRUN_CONFIRM_SPEND = "on"; +process.env.BLOCKRUN_CONFIRM_THRESHOLD = "0"; + +import { test, mock } from "node:test"; +import assert from "node:assert/strict"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import type { BudgetState } from "../src/types.js"; + +// Rail pin + traps, as in image-cost.test.ts. Every call below is DECLINED at +// the confirm dialog, so nothing past it may run — but the mocks make sure +// that if something did, it would fail here rather than pay. +const home = fs.mkdtempSync(path.join(os.tmpdir(), "blockrun-image-label-")); +const realHome = process.env.HOME; +const savedApiKey = process.env.BLOCKRUN_API_KEY; +process.env.HOME = home; +delete process.env.BLOCKRUN_API_KEY; +process.on("exit", () => { + if (realHome === undefined) delete process.env.HOME; else process.env.HOME = realHome; + if (savedApiKey === undefined) delete process.env.BLOCKRUN_API_KEY; else process.env.BLOCKRUN_API_KEY = savedApiKey; + fs.rmSync(home, { recursive: true, force: true }); +}); + +let networkCalls = 0; +const boom = () => { networkCalls++; throw new Error("UNEXPECTED_NETWORK_CALL"); }; +mock.module("../src/utils/wallet.js", { + namedExports: { + getApiBase: () => "https://blockrun.ai/api", + resolveGatewayUrl: (u: string) => u, + getChain: () => "base", + getImageClient: () => new Proxy({}, { get: () => boom }), + getOrCreateWalletKey: () => { throw new Error("label tests must not touch a wallet key"); }, + getWalletInfo: async () => ({ address: "0xTEST" }), + resolveSolanaKey: () => undefined, + }, +}); +mock.module("../src/utils/http.js", { + namedExports: { fetchWithTimeout: async () => boom(), isTimeoutError: () => false }, +}); + +const { registerImageTool, toImageDataUri, resolveImageRef } = await import("../src/tools/image.js"); + +// 1x1 transparent PNG +const PNG_BYTES = Buffer.from( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+M8AAAMBAQDJ/pLvAAAAAElFTkSuQmCC", + "base64", +); +const DATA_URI = `data:image/png;base64,${PNG_BYTES.toString("base64")}`; + +// Files live under a fresh tmp dir. On macOS os.tmpdir() is itself a symlink +// (/var -> /private/var), so realpath differs from the path we hand in — which +// is exactly the difference the label must show. +const dir = fs.mkdtempSync(path.join(os.tmpdir(), "blockrun-img-label-")); +const photo = path.join(dir, "photo.png"); +const logo = path.join(dir, "logo.jpg"); +const maskFile = path.join(dir, "mask.png"); +fs.writeFileSync(photo, PNG_BYTES); +fs.writeFileSync(logo, PNG_BYTES); +fs.writeFileSync(maskFile, PNG_BYTES); +const linkDir = fs.mkdtempSync(path.join(os.tmpdir(), "blockrun-img-link-")); +const innocentLink = path.join(linkDir, "cat.png"); +fs.symlinkSync(photo, innocentLink); +process.on("exit", () => { + fs.rmSync(dir, { recursive: true, force: true }); + fs.rmSync(linkDir, { recursive: true, force: true }); +}); +const real = (p: string) => fs.realpathSync(p); + +type Handler = (args: Record) => Promise<{ content: Array<{ type: string; text?: string }>; isError?: boolean }>; + +function harness() { + let handler: Handler | undefined; + const messages: string[] = []; + const server = { + registerTool: (_n: string, _c: unknown, h: Handler) => { handler = h; }, + server: { + getClientCapabilities: () => ({ elicitation: {} }), + elicitInput: async (req: { message: string }) => { messages.push(req.message); return { action: "decline" }; }, + }, + }; + const budget: BudgetState = { limit: null, spent: 0, calls: 0, agents: new Map() }; + registerImageTool(server as never, budget); + assert.ok(handler, "blockrun_image did not register a handler"); + networkCalls = 0; + return { + call: async (args: Record) => { + const res = await handler!(args); + return { res, text: res.content.map((p) => p.text ?? "").join("\n") }; + }, + budget, + messages, + // The confirm message's first line is "💸 BlockRun charge —