From 7c0e74586b333c7b5cbffa1b651846a9081502c5 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Fri, 25 Sep 2026 09:41:48 -0300 Subject: [PATCH 1/7] docs: make the user-nonce rule part of the application contract Every account starts at 0, validation accepts only the expected nonce, each included user op (business failures included) advances its sender's nonce by exactly one, nothing else changes a nonce, and an op at u32::MAX is never included. This was the wallet's behavior; the sequencer now relies on it to derive GET /nonce from persisted user ops instead of asking the engine, so the rule is stated in the contract, the trait docs, the C header, and AGENTS.md. --- AGENTS.md | 3 +- .../c-app-engine/include/application-engine.h | 15 ++++++--- docs/protocol/application-contract.md | 33 ++++++++++++++++--- examples/app-core/src/application/wallet.rs | 3 +- sequencer-core/src/application/mod.rs | 18 ++++++---- 5 files changed, 55 insertions(+), 17 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index d18eab3..c18315c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -212,6 +212,7 @@ Paths below are relative to `sequencer/src/`: - API validates the EIP-712 signature and enqueues a `SignedUserOp`. Method payload decoding happens during application execution, not at ingress. - **Deposits are direct-input-only** (L1 → L2) and must not be represented as user ops. - Rejections (`InvalidNonce`, `InvalidMaxFee`, `InsufficientFeeBalance`) produce no state mutation and are not persisted. These are protocol-level rejection semantics every app must implement: nonces prevent user-op replay, fees prevent spam against the sequencer's DA budget. ("Fee", not "gas" — the fee tracks DA; compute metering, if it ever exists, is a separate future concept.) +- **The user-nonce rule is contract, not wallet detail:** accounts start at 0, an op must carry exactly the expected nonce, each included op advances it by one, and direct inputs never touch it. `GET /nonce` derives from persisted user ops under this rule instead of asking the app ([application contract](docs/protocol/application-contract.md#user-nonces)). - Included txs are persisted as frame/batch data in `batches`, `frames`, `user_ops`, `safe_inputs`, and `application_inputs`. Recovery metadata lives in `safe_accepted_batches`; batch lifecycle state (sealed/invalidated) lives on the `batches` row itself as write-once timestamps. - Frame fee is persisted in `frames.fee` and is fixed for the lifetime of that frame. The next frame's fee is currently sampled from `batch_policy_derived.recommended_fee` at rotation; oracle bootstrap writes the price before any Tip can sample it, and `log_slack` applies the 10× margin in log space. This is present behavior, not a reason for the five-block clock policy; hoisting fee to the batch is a later design with its own trade-offs. - Wallet balances and nonces live in memory between checkpoints; restart restores @@ -239,7 +240,7 @@ Implementors of the `Application` trait must respect these contracts. The shared The sequencer persists every included user op and every ingested direct input. On restart, catch-up replays them in order against a fresh `Application` instance to rebuild state. **Any input that succeeded live must succeed on replay.** - `apply_direct_input` and `apply_valid_user_op` must not return `AppError::Internal` for any byte sequence that previously executed successfully. The canonical scheduler, catch-up, and recovery fold treat `Internal` as fatal: no canonical successor is defined. -- Validation returns `Accept`, `Reject(InvalidReason)`, or fatal `AppError`. A rejected op changes no state. Included business failures and malformed direct-input no-ops still advance progress; do not turn them into validation rejections. See the application contract for the wallet's nonce/fee semantics. +- Validation returns `Accept`, `Reject(InvalidReason)`, or fatal `AppError`. A rejected op changes no state. Included business failures and malformed direct-input no-ops still advance progress; do not turn them into validation rejections. See the application contract for the nonce rule and the wallet's fee semantics. - `validate_user_op` must be pure over the current app state. No side effects, no time dependence, no randomness. ### No implicit state diff --git a/bindings/c-app-engine/include/application-engine.h b/bindings/c-app-engine/include/application-engine.h index 15562c6..0512578 100644 --- a/bindings/c-app-engine/include/application-engine.h +++ b/bindings/c-app-engine/include/application-engine.h @@ -50,6 +50,12 @@ /// Validation acceptance, state transitions, progress, and outputs must agree with the canonical /// application. Rejection diagnostics and error messages need not be identical across builds. /// +/// Nonces follow one rule, which the host relies on to serve each sender's next nonce from the +/// ops it persisted: a genesis state starts every account at 0, validation accepts only the +/// sender's expected nonce, each executed user op advances its sender's nonce by exactly one, and +/// nothing else changes a nonce. An op at UINT32_MAX has no successor and is never executed. See +/// docs/protocol/application-contract.md. +/// /// Fees are uint16_t exponents with base 129/128, denominated in the fee token's smallest unit. /// The conversion is defined by sequencer-core/src/fee.rs and its build.rs-generated table: /// integer fixed-point arithmetic with 64 fractional bits, including its rounding and exponent @@ -142,8 +148,9 @@ typedef struct ApplicationEngineUserOp { } ApplicationEngineUserOp; /// @brief A user op that already passed validation, as the caller sequenced it. -/// @details Not the signed op: execution consumes the nonce the state expects, and the max fee -/// went with the guard the caller already settled, leaving the fee the frame charges. +/// @details Not the signed op: execution consumes the nonce the state expects, advancing it by +/// exactly one, and the max fee went with the guard the caller already settled, leaving the fee +/// the frame charges. typedef struct ApplicationEngineValidUserOp { ApplicationEngineEthereumAddress sender; ///< The recovered signer. uint16_t fee; ///< The charged frame fee exponent, base 129/128. @@ -296,7 +303,7 @@ APPLICATION_ENGINE_API ApplicationEngineStatus application_engine_validate_user_ const ApplicationEngineEthereumAddress *sender, const ApplicationEngineUserOp *user_op, uint16_t current_fee, ApplicationEngineInvalid *out_invalid) APPLICATION_ENGINE_NOEXCEPT; -/// @brief Execute a validated user op, consuming the current expected nonce. +/// @brief Execute a validated user op, advancing its sender's expected nonce by exactly one. /// @param engine The engine handle. /// @param user_op The validated op to execute. /// @param safe_block The covering frame safe block, folded into the clock as max(clock, it). @@ -319,7 +326,7 @@ APPLICATION_ENGINE_API ApplicationEngineStatus application_engine_execute_valid_ /// @param out_output_count How many outputs this input left waiting, written only on OK. /// @returns OK, IO_ERROR, or INTERNAL_ERROR. An error requires discarding the instance. /// @details An input the engine rejects is a counted no-op and still reports OK, the same way a -/// rejected user op does. Outputs behave as they do for a user op. +/// rejected user op does. Outputs behave as they do for a user op. It never changes a nonce. APPLICATION_ENGINE_API ApplicationEngineStatus application_engine_execute_direct_input(ApplicationEngine *engine, const ApplicationEngineDirectInput *input, uint64_t *out_output_count) APPLICATION_ENGINE_NOEXCEPT; diff --git a/docs/protocol/application-contract.md b/docs/protocol/application-contract.md index 24a4546..fd7b216 100644 --- a/docs/protocol/application-contract.md +++ b/docs/protocol/application-contract.md @@ -76,10 +76,11 @@ These outcomes have different protocol meanings: not persisted. `InvalidReason` covers nonce mismatch, insufficient max fee, and insufficient fee balance. - A successfully applied input is included even if the business operation - fails or is ignored. It advances progress. The wallet charges the fee and - consumes the nonce for a malformed method or failed transfer after - validation; malformed direct inputs are included no-ops. These semantics - must agree with the canonical application. + fails or is ignored. It advances progress, and an included user op consumes + its sender's nonce ([below](#user-nonces)). The wallet also charges the fee + for a malformed method or failed transfer after validation; malformed direct + inputs are included no-ops. These semantics must agree with the canonical + application. - `AppError`, whether `Internal` or `Io`, is fatal in validation and execution. It must not be disguised as a client rejection or an included no-op. Fatal here means discarding the engine: the host still distinguishes a @@ -90,6 +91,30 @@ Every input executed successfully live must execute successfully against the same prior state on replay. The sequencer persists included inputs and replays them on restart. It does not recover an instance after a failed hook. +#### User nonces + +Every application follows one nonce rule. The sequencer relies on it: +`GET /nonce` derives a sender's next nonce from the user ops it persisted, +without asking the engine. + +- A genesis state starts every account at nonce 0. +- Validation accepts an op only when its nonce equals the sender's expected + nonce. +- Each included user op, business failures included, advances its sender's + expected nonce by exactly one. Nothing else changes a nonce: not direct + inputs, not other senders' ops. +- Nonces are `u32`. An op carrying `u32::MAX` has no successor and is never + included. + +Under this rule a sender's expected nonce is one past its latest included op. +The exception is a sender with no op since the era baseline. A genesis +baseline puts it at 0, but a [rebuilt baseline](../recovery/cockroach.md) +carries nonces the rebuilt database has no ops for, so `GET /nonce` reads 0 +until that sender's first op. An engine outside the rule (gapped or keyed +nonces, a deposit that resets one) makes `GET /nonce` quote nonces the engine +rejects; one that includes an op at `u32::MAX` trips a fail-loud storage +invariant. + ### 3. The safe-block clock — `last_executed_safe_block` The clock is the maximum block carried by an executed input: frame `safe_block` diff --git a/examples/app-core/src/application/wallet.rs b/examples/app-core/src/application/wallet.rs index db7258f..58540b6 100644 --- a/examples/app-core/src/application/wallet.rs +++ b/examples/app-core/src/application/wallet.rs @@ -165,7 +165,8 @@ impl WalletApp { } // Wallet-specific read queries (not on the Application trait — the - // sequencer never asks; app-specific query surface belongs to the app). + // sequencer never asks: it derives nonces from persisted ops under the + // contract's nonce rule, and app-specific query surface belongs to the app). pub fn current_user_nonce(&self, sender: Address) -> u32 { self.expected_nonce(&sender) } diff --git a/sequencer-core/src/application/mod.rs b/sequencer-core/src/application/mod.rs index 65af049..80ac494 100644 --- a/sequencer-core/src/application/mod.rs +++ b/sequencer-core/src/application/mod.rs @@ -156,9 +156,10 @@ pub trait Application: Send + Sized { /// Zero permits only empty method payloads. fn max_method_payload_bytes() -> usize; - /// Pure validation predicate over current app state: nonce match - /// (user replay protection) and fee-balance coverage. Must not - /// mutate state. [`validate_and_execute_user_op`] enforces the protocol + /// Pure validation predicate over current app state: the op's nonce + /// equals the sender's expected nonce (user replay protection), and the + /// sender covers the fee. Must not mutate state. + /// [`validate_and_execute_user_op`] enforces the protocol /// `max_fee >= current_fee` guard before calling here. Rejection leaves /// the app unchanged; `AppError` is fatal and defines no successor. fn validate_user_op( @@ -169,8 +170,10 @@ pub trait Application: Send + Sized { ) -> Result; /// Apply a validated user op and advance progress exactly once on success, - /// using `safe_block` for the clock. Included business failures and no-ops - /// also advance progress. `AppError` is fatal: callers discard the instance. + /// using `safe_block` for the clock. Also advance the sender's nonce by + /// exactly one; the sequencer derives `GET /nonce` from this rule (see the + /// application contract). Included business failures and no-ops also + /// advance both. `AppError` is fatal: callers discard the instance. /// Execution callers use [`execute_valid_user_op`] to check the transition. fn apply_valid_user_op( &mut self, @@ -180,8 +183,9 @@ pub trait Application: Send + Sized { /// Apply a direct input and advance progress exactly once on success, /// using its L1 block number for the clock. Ignored or malformed inputs - /// still count. Execution callers use [`execute_direct_input`] to check - /// the transition; `AppError` requires discarding the instance. + /// still count. Never changes a user nonce. Execution callers use + /// [`execute_direct_input`] to check the transition; `AppError` requires + /// discarding the instance. fn apply_direct_input(&mut self, input: &DirectInput) -> Result; /// Return the progress embedded in the application's logical state. From 29fa9fb26d47ce76b7cdb67bc64e50cd21a13ae6 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Fri, 25 Sep 2026 09:41:58 -0300 Subject: [PATCH 2/7] feat: serve user nonces and the signing domain on ingress GET /nonce?sender= returns the nonce a sender signs next: one past its latest user op in a valid batch, or 0. It reads the rows the lane already writes in the FULL commit that authorizes each POST /tx ack, so a read after a 200 sees the op, and recovery lowers the value through the valid-batch filter. No new table or writer: a (sender, nonce) index on user_ops keeps the lookup off a full-history scan. POST /tx keeps no nonce or fee pre-filter; the api.rs module doc records why. GET /domain serves the EIP-712 domain /tx verifies against, keyed as eth_signTypedData_v4 expects, for clients to assert against the domain they pin. /fee and /nonce responses carry Cache-Control: no-store. The Rust SDK gains get_nonce and get_domain; GetFeeError becomes QueryError, shared by the three read routes. The README documents the one-op-in-flight client model and the known gap: a sender idle since a rebuilt baseline reads 0 until its first op. Closing that gap is a Track 6 follow-up. Schema: rewrites baseline migration 0001 (no deployed databases), so existing data directories need a fresh setup to get the index. --- AGENTS.md | 2 +- Cargo.lock | 2 + README.md | 40 +++- docs/plans/2026-07-coordination-tracks.md | 9 + docs/threat-model/README.md | 2 +- docs/watchdog/operator-deployment.md | 2 +- sdk/rust-client/Cargo.toml | 2 + sdk/rust-client/src/errors.rs | 23 +- sdk/rust-client/src/lib.rs | 51 ++++- sequencer-core/src/api.rs | 111 +++++++++- sequencer/src/commands/run/workers.rs | 2 +- sequencer/src/http.rs | 20 +- sequencer/src/ingress/api.rs | 200 ++++++++++++++++-- sequencer/src/ingress/mod.rs | 4 +- .../src/integration_tests/e2e_sequencer.rs | 61 ++++++ .../integration_tests/snapshot_endpoints.rs | 30 ++- .../src/integration_tests/ws_broadcaster.rs | 9 +- sequencer/src/storage/ingress.rs | 130 +++++++++++- .../src/storage/migrations/0001_schema.sql | 7 + 19 files changed, 630 insertions(+), 77 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index c18315c..b97d0f1 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -284,7 +284,7 @@ enforces write-once batch lifecycle, Tip uniqueness, and user-op identity. ## HTTP Endpoints -- **Ingress** (public-facing): `POST /tx`, `GET /fee`. +- **Ingress** (public-facing): `POST /tx`, `GET /fee`, `GET /nonce`, `GET /domain`. - **Egress** (internal indexers/watchdog): application-input subscriptions, snapshot/state downloads, and health probes. Snapshot/state endpoints have no authentication and **must not be exposed publicly**. Downloads hold a GC lease diff --git a/Cargo.lock b/Cargo.lock index 8c2d8d0..bfd831a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4324,8 +4324,10 @@ dependencies = [ name = "sequencer-rust-client" version = "0.1.0" dependencies = [ + "alloy-primitives", "reqwest 0.12.28", "sequencer-core", + "serde", "serde_json", "thiserror 2.0.19", "tokio", diff --git a/README.md b/README.md index 7d07793..0d3a04d 100644 --- a/README.md +++ b/README.md @@ -92,7 +92,7 @@ The sequencer is designed to handle: ### User Operations -Users submit signed operations via `POST /tx` (JSON). Operations are signed with EIP-712 using the rollup's chain ID and app address. The sequencer validates the signature, executes the operation against the current app state, and returns a soft confirmation. `GET /fee` quotes the live frame fee, the next-frame recommendation, and a suggested `max_fee` a wallet can sign. +Users submit signed operations via `POST /tx` (JSON). Operations are signed with EIP-712 using the rollup's chain ID and app address. The sequencer validates the signature, executes the operation against the current app state, and returns a soft confirmation. To sign, a wallet reads `GET /fee` (the live frame fee, the next-frame recommendation, and a suggested `max_fee`), `GET /nonce?sender=` (the nonce to sign next), and `GET /domain` (the EIP-712 domain, to assert against the one it pins). ### Sequenced Transaction Feed @@ -167,8 +167,9 @@ Most queue sizes, polling intervals, and safety limits are now internal runtime ## API -JSON `sender` fields in successful `POST /tx` responses and WebSocket messages -use EIP-55 checksum casing. Address fields in `/history` and `sender` fields in +JSON `sender` fields in successful `POST /tx` and `GET /nonce` responses and +WebSocket messages, and `verifyingContract` in `GET /domain`, use EIP-55 +checksum casing. Address fields in `/history` and `sender` fields in `/historical-l1-inputs` use lowercase hex. Clients must compare decoded 20-byte addresses and use one normalized encoding for account or projection keys across these routes. @@ -197,7 +198,7 @@ Notes: - payload size is bounded at ingress; oversized requests are rejected before entering the hot path. - overload is enforced at queue admission: if the inclusion-lane queue is full, `POST /tx` returns HTTP `429` with code `OVERLOADED` and message `queue full`. - queue capacity is an internal runtime constant tuned alongside inclusion-lane chunking to absorb short bursts; if this starts triggering persistently, it is a signal to revisit runtime sizing or throughput rather than add another admission layer. -- Browser wallets can call `POST /tx` and `GET /fee` from any origin with any request headers; preflight permits GET and POST and is cached for one hour. CORS is applied only to ingress. Egress routes remain operator-only and require network access controls. +- Browser wallets can call `POST /tx`, `GET /fee`, `GET /nonce`, and `GET /domain` from any origin with any request headers; preflight permits GET and POST and is cached for one hour. CORS is applied only to ingress. Egress routes remain operator-only and require network access controls. Success response after inclusion: @@ -222,9 +223,36 @@ Notes: - `fee` is frozen for the lifetime of the open frame (the live inclusion check). - `recommended_fee` is what the next frame will sample at rotation (currently after five newly-safe L1 blocks, best-effort). - `suggested_max_fee` is `max(fee, recommended_fee)` plus 1.5× log-space slack. Wallets can copy this into signed `max_fee`; the user pays the frame fee, not this cap. Clients that want their own policy can ignore it and combine the two facts themselves. -- `200` while an open frame exists (the admitted runtime always has one). +- `200` while an open frame exists (the admitted runtime always has one). Responses carry `Cache-Control: no-store`. - `503` with code `UNAVAILABLE` during shutdown, or if no open frame exists. +### `GET /nonce?sender=
` + +The nonce to sign into `sender`'s next user op. + +```json +{ "sender": "0xAbC...", "next_nonce": 7 } +``` + +Notes: + +- `sender` is parsed like the `POST /tx` field: `0x` plus 40 hex digits in any case, echoed in EIP-55 casing. A missing or malformed parameter is `400` with code `BAD_REQUEST`. +- `next_nonce` is one past the sender's latest included op, or 0 if it has none. It counts soft-confirmed ops: an op acknowledged with `200` before this request is counted. (The `nonce` in a `POST /tx` response or WS message is instead the nonce that op consumed.) +- Keep one op in flight per sender: sign `next_nonce`, submit, and increment on `200`. Concurrent submits from one sender are unsupported; they can reach the sequencer out of order and be rejected. After a submit times out, a `next_nonce` above the op's nonce means it was included; an unchanged one is inconclusive, since the op may still be queued, so resubmit the same signed op rather than signing a new one at that nonce. +- The value is a hint, not a reservation; the application is authoritative. It can go down, because a restart that runs automatic recovery may invalidate soft-confirmed ops. After a `422` bad-nonce rejection, query again rather than incrementing. +- After an operator rebuild from a checkpoint (`setup --recovery`), a sender with no op since the rebuild reads 0 even when its nonce in the rebuilt baseline is higher. A `422` bad-nonce rejection names the expected nonce (`bad nonce: expected N, got M`). +- Responses carry `Cache-Control: no-store`. `503` with code `UNAVAILABLE` during shutdown. + +### `GET /domain` + +The EIP-712 domain `POST /tx` verifies signatures against, keyed as `eth_signTypedData_v4` expects: + +```json +{ "name": "CartesiAppSequencer", "version": "1", "chainId": 31337, "verifyingContract": "0x..." } +``` + +Clients pin their own domain and assert that it matches this one, the way a wallet checks `eth_chainId`. Do not sign with a served domain unchecked: `chainId` and `verifyingContract` are what keep a signature from replaying on another deployment, and the name and version are the same everywhere. + ### `GET /ws/subscribe?era_id=&recovery_generation=&next_input=` WebSocket stream of the current application history, replaying from the inclusive @@ -450,7 +478,7 @@ They do not certify L1 freshness, submitter balance, or canonical agreement. - `examples/wallet-sequencer/`: binary crate composing the sequencer library with the placeholder wallet app - `sequencer/src/http.rs`: shared HTTP error type, JSON error shape, and `axum::serve` orchestration - `sequencer/src/runtime/`: process lock and shutdown scope; command bootstrap and config live in `commands/`, the shared clock in `clock.rs`, and EIP-712 domain construction in `sequencer-core/` -- `sequencer/src/ingress/`: public-facing — `POST /tx` and `GET /fee` (`api.rs`) and the inclusion lane (`inclusion_lane/`: hot-path loop, chunk/frame/batch rotation, catch-up, snapshot lifecycle) +- `sequencer/src/ingress/`: public-facing — `POST /tx`, `GET /fee`, `GET /nonce`, and `GET /domain` (`api.rs`) and the inclusion lane (`inclusion_lane/`: hot-path loop, chunk/frame/batch rotation, catch-up, snapshot lifecycle) - `sequencer/src/egress/`: internal read path — WS subscribe + health probes (`api/`) and the DB-backed ordered-L2Tx feed (`l2_tx_feed/`) - `sequencer/src/l1/`: L1 client surface — input reader, batch submitter, fee oracle, shared EIP-1559 estimation, provider, partition helper - `sequencer/src/recovery/`: preemptive recovery startup, runtime danger detector, mempool flusher diff --git a/docs/plans/2026-07-coordination-tracks.md b/docs/plans/2026-07-coordination-tracks.md index 7f591d7..873892b 100644 --- a/docs/plans/2026-07-coordination-tracks.md +++ b/docs/plans/2026-07-coordination-tracks.md @@ -94,6 +94,15 @@ Remaining checks need the actual consumer: - Decide output storage, checkpoint layout, ABI version negotiation, generated bindings, and linker policy from concrete engine requirements. The current drain protocol and path callback remain the contract until then. +- Make `GET /nonce` exact for senders idle since a rebuilt baseline; today + they read 0 ([application contract](../protocol/application-contract.md#user-nonces)). + Proposed: an `Application` nonce query plus an enumeration method. Setup + already holds the baseline engine when it publishes the baseline, so it can + enumerate it into a write-once `baseline_nonces` table in the same + transaction, the user-nonce counterpart of `batch_tree_anchor`. The point + query also lets the shared execution boundary check the one-step nonce + advance per op, as it checks progress. Both need C ABI exports and the + engine owner's agreement. Additional checkpoint primitives or asynchronous scheduling need a measured requirement. Any future microbatch priority scheme must preserve per-account diff --git a/docs/threat-model/README.md b/docs/threat-model/README.md index 6db808c..c6fd49b 100644 --- a/docs/threat-model/README.md +++ b/docs/threat-model/README.md @@ -25,7 +25,7 @@ What we are protecting: | Batch-submitter private key | Private | Held in operator infra. Not reachable by the network. | | Sequencer's own code | Trusted during normal operation | Tests/review prevent bugs; runtime invariant checks fail loud. Bugs require diagnosis and correction, followed by an operator rebuild if local state cannot be trusted. See "self-trust" below. | | **L1 mempool and block builders** | **Fully adversarial** | May reorder, delay, drop, or selectively include submitted transactions. Private mempools mean "dropped" is indistinguishable from "delayed indefinitely." | -| HTTP clients at `POST /tx` and `GET /fee` | Untrusted | Arbitrary public callers. May submit malformed, malicious, or replay payloads. `GET /fee` is an intentional public quote of the open-frame fee. | +| HTTP clients at `POST /tx`, `GET /fee`, `GET /nonce`, and `GET /domain` | Untrusted | Arbitrary public callers. May submit malformed, malicious, or replay payloads. The `GET` routes are intentional public reads: the open-frame fee, any sender's pending nonce (already public once its batches land on L1), and the signing domain. | | WebSocket subscribers at `/ws/subscribe` | Internal, but untrusted for data-exposure | Intended for internal indexers. Treat as public for what is exposed. | | Direct-input senders on L1 | Untrusted | Arbitrary L1 accounts calling InputBox. May submit any calldata. | diff --git a/docs/watchdog/operator-deployment.md b/docs/watchdog/operator-deployment.md index 10b8082..5ac8b2d 100644 --- a/docs/watchdog/operator-deployment.md +++ b/docs/watchdog/operator-deployment.md @@ -10,7 +10,7 @@ For **local development only** (Anvil + `sequencer-devnet`, CI smoke tests), use ```text ┌─────────────────────────────────────┐ - Internet / users │ Public ingress: POST /tx, GET /fee │ ← wallets + Internet / users │ Public ingress: /tx + read routes │ ← wallets └──────────────────┬──────────────────┘ │ ┌──────────────────▼──────────────────┐ diff --git a/sdk/rust-client/Cargo.toml b/sdk/rust-client/Cargo.toml index 10cf195..f3fc05a 100644 --- a/sdk/rust-client/Cargo.toml +++ b/sdk/rust-client/Cargo.toml @@ -10,8 +10,10 @@ readme = "../../README.md" authors.workspace = true [dependencies] +alloy-primitives = { workspace = true } sequencer-core = { path = "../../sequencer-core" } reqwest = { workspace = true, features = ["json"] } +serde = { workspace = true } serde_json = { workspace = true } thiserror = { workspace = true } tokio = { workspace = true } diff --git a/sdk/rust-client/src/errors.rs b/sdk/rust-client/src/errors.rs index cabe567..67ec3c3 100644 --- a/sdk/rust-client/src/errors.rs +++ b/sdk/rust-client/src/errors.rs @@ -57,14 +57,23 @@ pub enum SubmitRejected { Decode(String), } +/// A failed read from a public ingress query route: `/fee`, `/nonce`, or +/// `/domain`. #[derive(Debug, Error)] -pub enum GetFeeError { - #[error("fee request failed: {0}")] - Transport(#[from] SubmitTxError), - #[error("/fee rejected with status {status}: {body}")] - Http { status: u16, body: String }, - #[error("invalid /fee success body: {0}")] - Decode(String), +pub enum QueryError { + #[error("{route} request failed: {source}")] + Transport { + route: &'static str, + source: SubmitTxError, + }, + #[error("{route} rejected with status {status}: {body}")] + Http { + route: &'static str, + status: u16, + body: String, + }, + #[error("invalid {route} success body: {reason}")] + Decode { route: &'static str, reason: String }, } #[derive(Debug, Error)] diff --git a/sdk/rust-client/src/lib.rs b/sdk/rust-client/src/lib.rs index fefee4e..39552e4 100644 --- a/sdk/rust-client/src/lib.rs +++ b/sdk/rust-client/src/lib.rs @@ -5,7 +5,7 @@ mod errors; mod history; pub use errors::{ - ClientBuildError, GetFeeError, HistoryReadError, SnapshotError, SubmitRejected, SubmitTxError, + ClientBuildError, HistoryReadError, QueryError, SnapshotError, SubmitRejected, SubmitTxError, SubscribeError, }; @@ -18,7 +18,8 @@ pub use sequencer_core::history_api::{ HistoryBaseline, HistoryCompatibility, HistoryDeployment, HistoryInfo, }; -use sequencer_core::api::{FeeResponse, TxRequest, TxResponse}; +use alloy_primitives::Address; +use sequencer_core::api::{DomainResponse, FeeResponse, NonceResponse, TxRequest, TxResponse}; use std::time::Duration; use tokio::net::TcpStream; use tokio_tungstenite::{MaybeTlsStream, WebSocketStream, connect_async}; @@ -123,24 +124,53 @@ impl SequencerClient { serde_json::from_str::(&body).map_err(|e| SubmitRejected::Decode(e.to_string())) } - pub async fn get_fee(&self) -> Result { - let url = format!("{}/fee", self.endpoint.trim_end_matches('/')); + pub async fn get_fee(&self) -> Result { + self.get_json("/fee", "").await + } + + /// The nonce `sender` signs next. A hint for one op in flight: re-query + /// after a `422` bad-nonce rejection, since recovery can lower it. + pub async fn get_nonce(&self, sender: Address) -> Result { + self.get_json("/nonce", &format!("?sender={sender:#x}")) + .await + } + + /// The EIP-712 domain the sequencer verifies against. Compare it with a + /// pinned domain; never sign with it unchecked. + pub async fn get_domain(&self) -> Result { + self.get_json("/domain", "").await + } + + async fn get_json( + &self, + route: &'static str, + query: &str, + ) -> Result { + let url = format!("{}{route}{query}", self.endpoint.trim_end_matches('/')); + let transport = |source| QueryError::Transport { route, source }; let response = self .http_client .get(&url) .timeout(self.request_timeout) .send() .await - .map_err(map_reqwest_error)?; + .map_err(|e| transport(map_reqwest_error(e)))?; let status = response.status().as_u16(); let body = response .text() .await - .map_err(|e| SubmitTxError::IoRead(e.to_string()))?; + .map_err(|e| transport(SubmitTxError::IoRead(e.to_string())))?; if status != 200 { - return Err(GetFeeError::Http { status, body }); + return Err(QueryError::Http { + route, + status, + body, + }); } - serde_json::from_str::(&body).map_err(|e| GetFeeError::Decode(e.to_string())) + serde_json::from_str::(&body).map_err(|e| QueryError::Decode { + route, + reason: e.to_string(), + }) } /// Bounds response headers by the request timeout; callers own body cancellation. @@ -300,7 +330,10 @@ mod tests { .expect("fee request must retain its deadline independently of snapshot streaming"); assert!(matches!( result, - Err(GetFeeError::Transport(SubmitTxError::TimeoutRead)) + Err(QueryError::Transport { + source: SubmitTxError::TimeoutRead, + .. + }) )); } diff --git a/sequencer-core/src/api.rs b/sequencer-core/src/api.rs index 7650648..0404eb8 100644 --- a/sequencer-core/src/api.rs +++ b/sequencer-core/src/api.rs @@ -34,7 +34,7 @@ impl TxRequest { self.validate_payload_size(max_user_op_data_bytes)?; let signature = self.decode_signature()?; - let expected_sender = self.decode_address()?; + let expected_sender = parse_sender_address(&self.sender)?; let recovered_sender = recover_sender(&self.message, &signature, domain)?; if expected_sender != recovered_sender { @@ -83,14 +83,20 @@ impl TxRequest { } parse_signature(&signature_bytes) } +} - fn decode_address(&self) -> Result { - let bytes = decode_hex_0x(self.sender.as_str()).map_err(TxRequestError::bad_request)?; - if bytes.len() != Self::ADDRESS_BYTES { - return Err(TxRequestError::bad_request("address must be 20 bytes")); - } - Ok(Address::from_slice(&bytes)) +/// Parse a sender the way `POST /tx` does: `0x` plus 40 hex digits in any +/// letter case. EIP-55 checksums are not enforced; the signature, not the +/// casing, binds a sender. +pub fn parse_sender_address(value: &str) -> Result { + if value.len() != TxRequest::ADDRESS_HEX_LEN { + return Err(TxRequestError::bad_request(format!( + "sender must be {} hex chars (0x + 20 bytes)", + TxRequest::ADDRESS_HEX_LEN + ))); } + let bytes = decode_hex_0x(value).map_err(TxRequestError::bad_request)?; + Ok(Address::from_slice(&bytes)) } #[derive(Debug, Error, Clone)] @@ -142,6 +148,44 @@ impl FeeResponse { } } +/// `GET /nonce` body. `next_nonce` is the value to sign into `UserOp.nonce`; +/// [`TxResponse::nonce`] is instead the nonce an included op consumed. +/// `sender` is echoed in EIP-55 casing, like `POST /tx` responses. +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] +pub struct NonceResponse { + pub sender: String, + pub next_nonce: u32, +} + +/// `GET /domain` body: the EIP-712 domain `POST /tx` verifies signatures +/// against, keyed as `eth_signTypedData_v4` expects. Clients pin their own +/// domain and assert it matches, as a wallet checks `eth_chainId`. +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] +#[serde(rename_all = "camelCase")] +pub struct DomainResponse { + pub name: String, + pub version: String, + pub chain_id: u64, + /// EIP-55 casing. + pub verifying_contract: String, +} + +impl DomainResponse { + /// `None` unless the domain has exactly the shape + /// [`crate::build_input_domain`] produces: all four fields, no salt. + pub fn from_domain(domain: &Eip712Domain) -> Option { + if domain.salt.is_some() { + return None; + } + Some(Self { + name: domain.name.as_deref()?.to_owned(), + version: domain.version.as_deref()?.to_owned(), + chain_id: u64::try_from(domain.chain_id?).ok()?, + verifying_contract: domain.verifying_contract?.to_string(), + }) + } +} + pub type WsTxMessage = BroadcastTxMessage; fn decode_hex_0x(value: &str) -> Result, String> { @@ -178,3 +222,56 @@ fn parse_signature(bytes: &[u8]) -> Result { } }) } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn sender_parser_accepts_any_case_and_nothing_else() { + let address = Address::repeat_byte(0xab); + let lower = format!("{address:#x}"); + for accepted in [lower.clone(), lower.to_uppercase().replacen("0X", "0x", 1)] { + assert_eq!(parse_sender_address(&accepted).unwrap(), address); + } + assert_eq!( + parse_sender_address(&address.to_checksum(None)).unwrap(), + address + ); + for rejected in [ + lower.trim_start_matches("0x"), + &lower[..lower.len() - 2], + &format!("{lower}00"), + &lower.replacen('a', "g", 1), + &lower.replacen("0x", "0X", 1), + ] { + assert!( + parse_sender_address(rejected).is_err(), + "accepted {rejected:?}" + ); + } + } + + #[test] + fn domain_response_uses_eip712_keys_and_requires_the_full_domain() { + let app = Address::repeat_byte(0xab); + let domain = crate::build_input_domain(31337, app); + let served = DomainResponse::from_domain(&domain).unwrap(); + assert_eq!( + serde_json::to_value(&served).unwrap(), + serde_json::json!({ + "name": crate::DOMAIN_NAME, + "version": crate::DOMAIN_VERSION, + "chainId": 31337, + "verifyingContract": app.to_checksum(None), + }) + ); + + let mut partial = domain.clone(); + partial.verifying_contract = None; + assert_eq!(DomainResponse::from_domain(&partial), None); + let mut salted = domain; + salted.salt = Some(alloy_primitives::B256::ZERO); + assert_eq!(DomainResponse::from_domain(&salted), None); + } +} diff --git a/sequencer/src/commands/run/workers.rs b/sequencer/src/commands/run/workers.rs index 9fde839..eab65d2 100644 --- a/sequencer/src/commands/run/workers.rs +++ b/sequencer/src/commands/run/workers.rs @@ -293,7 +293,7 @@ impl PreparedRuntime { let submitter = submitter.start_preflighted(shutdown.clone()); let detector = detector.start_preflighted(shutdown.signal()); let fee_oracle = fee_oracle.map(|oracle| oracle.start(shutdown.signal())); - // HTTP server (ingress /tx + /fee + egress /ws/subscribe + /health, currently merged). + // HTTP server (ingress /tx + read routes, egress /ws/subscribe + /health, currently merged). let server = http::start_on_listener( listener, tx, diff --git a/sequencer/src/http.rs b/sequencer/src/http.rs index 44284db..1a0369a 100644 --- a/sequencer/src/http.rs +++ b/sequencer/src/http.rs @@ -2,8 +2,9 @@ // SPDX-License-Identifier: Apache-2.0 (see LICENSE) //! Shared HTTP surface: error type + JSON response shape used by both -//! ingress (`/tx`, `/fee`) and egress (`/ws/subscribe`, future routes), plus the -//! `axum::serve` orchestration that wires the two side routers together. +//! ingress (`/tx`, `/fee`, `/nonce`, `/domain`) and egress (`/ws/subscribe`, +//! future routes), plus the `axum::serve` orchestration that wires the two +//! side routers together. //! //! Today both sides serve from one listener; the planned api split puts each //! side on its own port (same binary, two listeners). When that lands, the @@ -26,11 +27,11 @@ use tower_http::trace::TraceLayer; pub use crate::egress::api::SnapshotState; use crate::egress::api::SubscribeState; use crate::egress::l2_tx_feed::L2TxFeed; -use crate::ingress::api::{FeeState, SubmitState}; +use crate::ingress::api::{ReadState, SubmitState}; use crate::ingress::inclusion_lane::{PendingUserOp, SequencerError}; use crate::runtime::shutdown::{RuntimeScope, abort_terminal}; use crate::storage::ReleaseScheduler; -use sequencer_core::api::{TxRequest, TxRequestError}; +use sequencer_core::api::{DomainResponse, TxRequest, TxRequestError}; #[derive(Debug, Error, Clone)] pub enum ApiError { @@ -211,7 +212,7 @@ async fn run_snapshot_release_supervisor( // ── Blocking storage tasks ───────────────────────────────────────────────── // -// Shared by ingress (`GET /fee`) and egress snapshot handlers. Classify +// Shared by ingress (`GET /fee`, `GET /nonce`) and egress snapshot handlers. Classify // inside the blocking task: cancellation of the HTTP request must not // discard a persistent fault discovered by work that already started. @@ -307,14 +308,19 @@ pub(crate) fn start_on_listener( tx_sender: tx_sender.clone(), shutdown: shutdown.clone(), }); + // Serve exactly the domain `/tx` verifies against. Production builds it + // with `build_input_domain`, which always sets all four fields. + let served_domain = DomainResponse::from_domain(&config.domain) + .expect("API domain must carry name, version, chainId, and verifyingContract"); let submit_state = Arc::new(SubmitState::new( tx_sender, config.domain, config.max_user_op_data_bytes, shutdown.clone(), )); - let fee_state = Arc::new(FeeState::new( + let read_state = Arc::new(ReadState::new( snapshot_state.db_path.clone(), + served_domain, shutdown.clone(), )); let subscribe_state = Arc::new(SubscribeState::new( @@ -322,7 +328,7 @@ pub(crate) fn start_on_listener( tx_feed, config.ws_max_subscribers, )); - let app: Router = crate::ingress::api::router(submit_state, fee_state) + let app: Router = crate::ingress::api::router(submit_state, read_state) .merge(crate::egress::api::router( subscribe_state, health_state, diff --git a/sequencer/src/ingress/api.rs b/sequencer/src/ingress/api.rs index 1575334..e88e574 100644 --- a/sequencer/src/ingress/api.rs +++ b/sequencer/src/ingress/api.rs @@ -8,14 +8,27 @@ //! perspective: 200 means included. //! - `GET /fee` — quote the open-frame fee, recommended fee, and a suggested //! `max_fee` so a wallet can sign before submitting. +//! - `GET /nonce?sender=` — the nonce `sender` must sign next, derived from the +//! user ops the lane has committed. +//! - `GET /domain` — the EIP-712 domain signatures are verified against. +//! +//! `/fee` and `/nonce` query SQLite on a read-only connection, never the lane: +//! WAL readers do not block its writes. +//! +//! Admission checks only what the lane cannot cheaply check itself: payload +//! size and the signature. Nonce and `max_fee` are left to the lane. A stale +//! op costs it an in-memory comparison and no transaction, whereas an ingress +//! pre-check would add a SQLite read to every honest submit, and a future +//! nonce or a fresh address bypasses it. Revisit if persistent `429`s trace +//! back to stale-nonce floods. use std::sync::Arc; use std::time::{Duration, SystemTime}; use alloy_sol_types::Eip712Domain; use axum::Router; -use axum::extract::{Json, State}; -use axum::http::{Method, StatusCode}; +use axum::extract::{Json, Query, State}; +use axum::http::{HeaderValue, Method, StatusCode, header}; use axum::response::{IntoResponse, Response}; use axum::routing::{get, post}; use tokio::sync::mpsc::{self, error::TrySendError}; @@ -27,8 +40,11 @@ use crate::http::{ApiError, storage_task}; use crate::ingress::inclusion_lane::PendingUserOp; use crate::runtime::shutdown::RuntimeScope; use crate::storage::Storage; -use sequencer_core::api::{FeeResponse, TxRequest, TxResponse}; +use sequencer_core::api::{ + DomainResponse, FeeResponse, NonceResponse, TxRequest, TxResponse, parse_sender_address, +}; use sequencer_core::user_op::SignedUserOp; +use serde::Deserialize; /// State for the submit endpoint. Kept narrow — only what `/tx` actually needs. #[derive(Clone)] @@ -63,17 +79,23 @@ impl SubmitState { } } -/// State for `GET /fee`. Reads the open-frame fee from SQLite; the inclusion -/// lane remains the sole writer of that fact. +/// State for the public read routes. `/fee` reads the open frame and the fee +/// policy; `/nonce` reads lane-written user ops filtered by batch validity; +/// `/domain` is fixed for the process. #[derive(Clone)] -pub(crate) struct FeeState { +pub(crate) struct ReadState { db_path: String, + domain: DomainResponse, shutdown: RuntimeScope, } -impl FeeState { - pub(crate) fn new(db_path: String, shutdown: RuntimeScope) -> Self { - Self { db_path, shutdown } +impl ReadState { + pub(crate) fn new(db_path: String, domain: DomainResponse, shutdown: RuntimeScope) -> Self { + Self { + db_path, + domain, + shutdown, + } } fn reject_if_shutting_down(&self) -> Result<(), ApiError> { @@ -86,11 +108,17 @@ impl FeeState { } /// Build the ingress router. Caller wires it into an `axum::serve` listener. -pub(crate) fn router(submit: Arc, fee: Arc) -> Router { +pub(crate) fn router(submit: Arc, read: Arc) -> Router { Router::new() .route("/tx", post(submit_tx)) .with_state(submit) - .merge(Router::new().route("/fee", get(get_fee)).with_state(fee)) + .merge( + Router::new() + .route("/fee", get(get_fee)) + .route("/nonce", get(get_nonce)) + .route("/domain", get(get_domain)) + .with_state(read), + ) .layer( CorsLayer::new() .allow_origin(Any) @@ -127,7 +155,7 @@ async fn submit_tx( .into_response()) } -async fn get_fee(State(state): State>) -> Result { +async fn get_fee(State(state): State>) -> Result { state.reject_if_shutting_down()?; let db_path = state.db_path.clone(); let result = storage_task(state.shutdown.clone(), "read fee quote", move |_scope| { @@ -137,7 +165,7 @@ async fn get_fee(State(state): State>) -> Result { - Ok(Json(FeeResponse::quote(fee, recommended_fee)).into_response()) + Ok(no_store(Json(FeeResponse::quote(fee, recommended_fee)))) } Ok(None) => Err(ApiError::unavailable("no open frame")), Err(err) => { @@ -147,6 +175,58 @@ async fn get_fee(State(state): State>) -> Result>, + query: Result, axum::extract::rejection::QueryRejection>, +) -> Result { + state.reject_if_shutting_down()?; + let sender = query + .ok() + .and_then(|Query(query)| parse_sender_address(&query.sender).ok()) + .ok_or_else(|| ApiError::bad_request(INVALID_NONCE_QUERY))?; + let db_path = state.db_path.clone(); + let result = storage_task( + state.shutdown.clone(), + "read next user nonce", + move |_scope| { + let mut storage = Storage::open_read_only(&db_path)?; + Ok(storage.next_user_nonce(sender)?) + }, + ) + .await; + match result { + Ok(next_nonce) => Ok(no_store(Json(NonceResponse { + sender: sender.to_string(), + next_nonce, + }))), + Err(err) => { + tracing::warn!(error = %err, "GET /nonce failed"); + Err(ApiError::internal_error("nonce unavailable")) + } + } +} + +async fn get_domain(State(state): State>) -> Json { + Json(state.domain.clone()) +} + +/// Quotes go stale within blocks and nonces can go down after recovery, so +/// intermediaries must not serve them from cache. +fn no_store(body: impl IntoResponse) -> Response { + ( + [(header::CACHE_CONTROL, HeaderValue::from_static("no-store"))], + body, + ) + .into_response() +} + /// Normalize JSON-extractor failures into fixed client-facing messages. /// Keeps the public API contract stable across axum upgrades and avoids /// reflecting parser internals (serde line/column, token excerpts) to callers. @@ -279,7 +359,7 @@ mod tests { async fn get_fee_rejects_when_shutdown_has_started() { let shutdown = RuntimeScope::default(); shutdown.request_shutdown(); - let state = Arc::new(FeeState::new("unused.db".into(), shutdown)); + let state = read_state("unused.db".into(), shutdown); let err = get_fee(State(state)) .await @@ -293,10 +373,10 @@ mod tests { let db = TempDir::new().expect("create temp dir"); let db_path = db.path().join("sequencer.db"); let _storage = Storage::open(&db_path.to_string_lossy()).expect("create db"); - let state = Arc::new(FeeState::new( + let state = read_state( db_path.to_string_lossy().into_owned(), RuntimeScope::default(), - )); + ); let err = get_fee(State(state)) .await @@ -329,10 +409,10 @@ mod tests { .expect("inject impossible policy row"); drop(conn); - let state = Arc::new(FeeState::new( + let state = read_state( db_path.to_string_lossy().into_owned(), RuntimeScope::default(), - )); + ); let result = get_fee(State(state)).await; panic!( "terminal fee fault returned instead of aborting: {:?}", @@ -340,6 +420,90 @@ mod tests { ); } + fn read_state(db_path: String, shutdown: RuntimeScope) -> Arc { + let domain = sequencer_core::build_input_domain(31337, Address::repeat_byte(0xab)); + Arc::new(ReadState::new( + db_path, + DomainResponse::from_domain(&domain).expect("complete domain"), + shutdown, + )) + } + + fn nonce_query( + uri: &str, + ) -> Result, axum::extract::rejection::QueryRejection> { + Query::try_from_uri(&uri.parse().expect("test URI")) + } + + async fn json_body(response: Response) -> serde_json::Value { + let bytes = axum::body::to_bytes(response.into_body(), usize::MAX) + .await + .expect("read body"); + serde_json::from_slice(&bytes).expect("JSON body") + } + + #[tokio::test(flavor = "current_thread")] + async fn get_nonce_for_an_unseen_sender_is_zero_and_uncacheable() { + let db = TempDir::new().expect("create temp dir"); + let db_path = db.path().join("sequencer.db"); + let _storage = Storage::open(&db_path.to_string_lossy()).expect("create db"); + let state = read_state( + db_path.to_string_lossy().into_owned(), + RuntimeScope::default(), + ); + let sender = Address::repeat_byte(0xcd); + let lowercase = format!("{sender:#x}"); + + let response = get_nonce( + State(state), + nonce_query(&format!("/nonce?sender={lowercase}")), + ) + .await + .expect("an unseen sender is an ordinary answer, not a storage fault"); + assert_eq!(response.status(), StatusCode::OK); + assert_eq!( + response.headers().get(header::CACHE_CONTROL).unwrap(), + "no-store" + ); + assert_eq!( + json_body(response).await, + serde_json::json!({ "sender": sender.to_checksum(None), "next_nonce": 0 }) + ); + } + + #[tokio::test(flavor = "current_thread")] + async fn get_nonce_rejects_a_missing_or_malformed_sender() { + let state = read_state("unused.db".into(), RuntimeScope::default()); + for uri in [ + "/nonce", + "/nonce?address=0x0000000000000000000000000000000000000000", + "/nonce?sender=0000000000000000000000000000000000000000", + "/nonce?sender=0x00", + "/nonce?sender=0xzz00000000000000000000000000000000000000", + ] { + let err = get_nonce(State(state.clone()), nonce_query(uri)) + .await + .expect_err(uri); + assert_eq!(err.status(), StatusCode::BAD_REQUEST, "{uri}"); + assert_eq!(err.to_string(), INVALID_NONCE_QUERY, "{uri}"); + } + } + + #[tokio::test(flavor = "current_thread")] + async fn get_nonce_rejects_when_shutdown_has_started() { + let shutdown = RuntimeScope::default(); + shutdown.request_shutdown(); + let state = read_state("unused.db".into(), shutdown); + + let err = get_nonce( + State(state), + nonce_query("/nonce?sender=0x0000000000000000000000000000000000000000"), + ) + .await + .expect_err("nonce should be rejected during shutdown"); + assert_eq!(err.status(), StatusCode::SERVICE_UNAVAILABLE); + } + fn sign_user_op_hex( domain: &Eip712Domain, user_op: &UserOp, diff --git a/sequencer/src/ingress/mod.rs b/sequencer/src/ingress/mod.rs index 3c64a32..caaf964 100644 --- a/sequencer/src/ingress/mod.rs +++ b/sequencer/src/ingress/mod.rs @@ -1,8 +1,8 @@ // (c) Cartesi and individual authors (see AUTHORS) // SPDX-License-Identifier: Apache-2.0 (see LICENSE) -//! Inbound side: public HTTP (`POST /tx`, `GET /fee`) and the inclusion lane -//! that consumes the submit queue. The lane is the only writer of open +//! Inbound side: public HTTP (`POST /tx`, `GET /fee`, `GET /nonce`, +//! `GET /domain`) and the inclusion lane that consumes the submit queue. The lane is the only writer of open //! batch/frame state in storage. pub mod api; diff --git a/sequencer/src/integration_tests/e2e_sequencer.rs b/sequencer/src/integration_tests/e2e_sequencer.rs index 8b87149..0011a6e 100644 --- a/sequencer/src/integration_tests/e2e_sequencer.rs +++ b/sequencer/src/integration_tests/e2e_sequencer.rs @@ -880,6 +880,10 @@ async fn api_quotes_open_frame_fee() { quoted.fee, 1356, "quoted fee must match the bootstrapped open-frame fee" ); + let raw = reqwest::get(format!("http://{}/fee", runtime.addr)) + .await + .expect("raw GET /fee"); + assert_eq!(raw.headers()["cache-control"], "no-store"); assert_eq!( quoted.recommended_fee, 1356, "bootstrapped recommended_fee matches the open-frame fee" @@ -893,6 +897,63 @@ async fn api_quotes_open_frame_fee() { shutdown_runtime(runtime).await; } +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn api_serves_the_next_nonce_after_each_ack_and_the_signing_domain() { + let db = temp_db("nonce-endpoint"); + let domain = test_domain(); + let signing_key = SigningKey::from_bytes((&[11_u8; 32]).into()).expect("create signing key"); + let sender = address_from_signing_key(&signing_key); + bootstrap_open_frame_with_deposits(db.path.as_str(), &[(sender, U256::from(1_000_000_u64))]); + + let Some(runtime) = start_full_server(db.path.as_str(), domain.clone()).await else { + return; + }; + let endpoint = format!("http://{}", runtime.addr); + let client = SequencerClient::new_with_timeout(endpoint, Duration::from_secs(2)) + .expect("build sequencer client"); + + // Sign only with a domain rebuilt from the served JSON: /tx accepting those + // signatures proves /domain is the domain it verifies against. + let served = client.get_domain().await.expect("GET /domain"); + let domain = Eip712Domain { + name: Some(served.name.into()), + version: Some(served.version.into()), + chain_id: Some(U256::from(served.chain_id)), + verifying_contract: Some(served.verifying_contract.parse().expect("served address")), + salt: None, + }; + + for nonce in 0..2 { + let quoted = client.get_nonce(sender).await.expect("GET /nonce"); + assert_eq!(quoted.sender, sender.to_checksum(None)); + assert_eq!( + quoted.next_nonce, nonce, + "a read after the previous 200 must see that op" + ); + let user_op = UserOp { + nonce, + max_fee: 1356, + data: ssz::Encode::as_ssz_bytes(&Method::Withdrawal(Withdrawal { + amount: U256::from(0_u64), + })) + .into(), + }; + let request = TxRequest { + signature: sign_user_op_hex(&domain, &user_op, &signing_key), + sender: sender.to_string(), + message: user_op, + }; + let (status, body) = client + .submit_tx_with_status(&request) + .await + .expect("submit tx"); + assert_eq!(status, 200, "nonce {nonce}: {body}"); + } + assert_eq!(client.get_nonce(sender).await.unwrap().next_nonce, 2); + + shutdown_runtime(runtime).await; +} + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn api_rejects_user_op_when_balance_below_fee_cost() { // if sender's balance < `fee_to_linear(current_frame_fee)` the diff --git a/sequencer/src/integration_tests/snapshot_endpoints.rs b/sequencer/src/integration_tests/snapshot_endpoints.rs index 05d3c65..7eaff6b 100644 --- a/sequencer/src/integration_tests/snapshot_endpoints.rs +++ b/sequencer/src/integration_tests/snapshot_endpoints.rs @@ -28,13 +28,7 @@ use tokio::sync::mpsc; use super::common::temp_db; fn dummy_domain() -> Eip712Domain { - Eip712Domain { - name: None, - version: None, - chain_id: None, - verifying_contract: None, - salt: None, - } + sequencer_core::build_input_domain(1, alloy_primitives::Address::ZERO) } /// Holds the server task + channel alive for the duration of a test. @@ -817,6 +811,28 @@ async fn cors_is_limited_to_ingress_and_covers_rejections() { .expect("GET /fee"); assert_eq!(fee.headers()["access-control-allow-origin"], "*"); + for (route, status) in [ + ( + "/nonce?sender=0x00000000000000000000000000000000000000aa", + 200, + ), + ("/nonce?sender=0xaa", 400), + ("/domain", 200), + ] { + let response = client + .get(server.url(route)) + .header("Origin", "https://wallet.example") + .send() + .await + .expect("ingress read"); + assert_eq!(response.status().as_u16(), status, "{route}"); + assert_eq!( + response.headers()["access-control-allow-origin"], + "*", + "{route}" + ); + } + for route in ["/livez", "/finalized_state"] { let response = client .get(server.url(route)) diff --git a/sequencer/src/integration_tests/ws_broadcaster.rs b/sequencer/src/integration_tests/ws_broadcaster.rs index 04d3ca9..a057e93 100644 --- a/sequencer/src/integration_tests/ws_broadcaster.rs +++ b/sequencer/src/integration_tests/ws_broadcaster.rs @@ -6,7 +6,6 @@ use std::time::{Duration, SystemTime}; use crate::storage::{ApplicationInputRow, L2TxContext}; use alloy_primitives::{Address, Signature}; -use alloy_sol_types::Eip712Domain; use app_core::application::MAX_METHOD_PAYLOAD_BYTES; use futures_util::{SinkExt, StreamExt}; use sequencer::egress::l2_tx_feed::{L2TxFeed, L2TxFeedConfig}; @@ -423,13 +422,7 @@ async fn start_test_server_with_limits( ApiConfig { ws_max_subscribers, ..ApiConfig::new( - Eip712Domain { - name: None, - version: None, - chain_id: None, - verifying_contract: None, - salt: None, - }, + sequencer_core::build_input_domain(1, alloy_primitives::Address::ZERO), MAX_METHOD_PAYLOAD_BYTES, ) }, diff --git a/sequencer/src/storage/ingress.rs b/sequencer/src/storage/ingress.rs index 2d7c090..e39a6c9 100644 --- a/sequencer/src/storage/ingress.rs +++ b/sequencer/src/storage/ingress.rs @@ -6,7 +6,9 @@ //! //! The lane also reads classified external directs and the open //! state (resumed on startup) — those reads live here too because they're driven -//! by the lane's flow, not by an L1 ingress event. +//! by the lane's flow, not by an L1 ingress event. The public ingress reads +//! live here too: the fee quote and the next user nonce, derived from the +//! lane's user-op rows. use std::path::Path; @@ -18,7 +20,8 @@ use super::StoredSafeInput; #[cfg(test)] use super::convert::external_u64_to_i64; use super::convert::{ - from_unix_ms, i64_to_u64, now_unix_ms, saturating_query_bound, to_unix_ms, u64_to_i64, + from_unix_ms, i64_to_u32, i64_to_u64, now_unix_ms, saturating_query_bound, to_unix_ms, + u64_to_i64, }; use super::history::{next_executed_input_count_in, query_history_state}; use super::mutations::{ @@ -37,6 +40,12 @@ use super::{ }; use crate::ingress::inclusion_lane::{IncludedUserOp, PendingUserOp}; +const LAST_VALID_USER_NONCE_SQL: &str = "SELECT u.nonce FROM user_ops u + WHERE u.sender = ?1 + AND EXISTS (SELECT 1 FROM valid_batches b WHERE b.batch_index = u.batch_index) + ORDER BY u.nonce DESC + LIMIT 1"; + impl Storage { /// First L1 input beyond the latest surviving frame's accounted block, /// bounded below by the immutable era prefix. @@ -61,6 +70,26 @@ impl Storage { }) } + /// The nonce `sender` must sign next: one past its latest user op in a + /// valid batch, or 0 without one. Exact under the application contract's + /// nonce rule, except that a sender idle since a rebuilt baseline reads 0. + /// + /// Invalidated batches keep their rows, and resubmission after a cascade + /// reuses their nonces, so the validity filter is load-bearing. + pub fn next_user_nonce(&mut self, sender: Address) -> Result { + self.read(|tx| { + let last_nonce: Option = tx + .prepare_cached(LAST_VALID_USER_NONCE_SQL)? + .query_row(params![sender.as_slice()], |row| row.get(0)) + .optional()?; + Ok(last_nonce.map_or(0, |nonce| { + i64_to_u32(nonce) + .checked_add(1) + .expect("included user nonce u32::MAX has no successor: contract-impossible") + })) + }) + } + /// Bootstrap the very first batch + frame with explicit values, returning /// its loaded [`WriteHead`]. Asserts no open state exists. /// @@ -1539,4 +1568,101 @@ mod tests { storage.close_frame_and_batch(&mut head, 10).unwrap(); assert_eq!(storage.next_undrained_safe_input_index().unwrap(), 1); } + + fn included_user_op_from(sender: Address, nonce: u32, offset: u64) -> IncludedUserOp { + let mut included = included_user_op(nonce, offset); + included.pending.signed.sender = sender; + included + } + + #[test] + fn next_user_nonce_follows_the_latest_valid_op_across_a_cascade() { + let alice = Address::repeat_byte(0xa1); + let bob = Address::repeat_byte(0xb0); + let db = temp_db("next-user-nonce"); + let mut storage = Storage::open(&db.path).unwrap(); + let mut head = storage + .initialize_open_state(0, SafeInputRange::empty_at(0)) + .unwrap(); + assert_eq!(storage.next_user_nonce(alice).unwrap(), 0, "unseen sender"); + + storage + .append_executed_user_ops_chunk( + &mut head, + &[ + included_user_op_from(alice, 0, 0), + included_user_op_from(alice, 1, 1), + included_user_op_from(bob, 0, 2), + ], + ) + .unwrap(); + storage.close_frame_and_batch(&mut head, 0).unwrap(); + storage + .append_executed_user_ops_chunk( + &mut head, + &[ + included_user_op_from(alice, 2, 3), + included_user_op_from(alice, 3, 4), + ], + ) + .unwrap(); + assert_eq!(storage.next_user_nonce(alice).unwrap(), 4); + assert_eq!(storage.next_user_nonce(bob).unwrap(), 1); + + // A cascade rolls Alice back; her invalidated rows stay for audit. + storage.insert_invalid_batch(head.batch_index).unwrap(); + assert_eq!(storage.next_user_nonce(alice).unwrap(), 2); + assert_eq!(storage.next_user_nonce(bob).unwrap(), 1); + + // Resubmitting nonce 2 leaves a duplicate (sender, nonce) pair across + // the invalid and valid batches; only the valid one counts. + storage + .append_safe_inputs(0, &[], SENDER_A, &default_protocol_timing()) + .unwrap(); + storage.ensure_open_tip().unwrap(); + let mut head = storage.open_state().unwrap().unwrap(); + let offset = storage.next_executed_input_count().unwrap().get(); + storage + .append_executed_user_ops_chunk(&mut head, &[included_user_op_from(alice, 2, offset)]) + .unwrap(); + assert_eq!(storage.next_user_nonce(alice).unwrap(), 3); + let raw_max: i64 = storage + .conn + .query_row( + "SELECT MAX(nonce) FROM user_ops WHERE sender = ?1", + [alice.as_slice()], + |row| row.get(0), + ) + .unwrap(); + assert_eq!( + raw_max, 3, + "an unfiltered MAX would still report the rolled-back op" + ); + } + + #[test] + fn next_user_nonce_seeks_the_sender_index_without_sorting() { + let db = temp_db("next-user-nonce-plan"); + let storage = Storage::open(&db.path).unwrap(); + let plan: Vec = storage + .conn + .prepare(&format!( + "EXPLAIN QUERY PLAN {}", + super::LAST_VALID_USER_NONCE_SQL + )) + .unwrap() + .query_map([Address::ZERO.as_slice()], |row| row.get(3)) + .unwrap() + .collect::>() + .unwrap(); + let plan = plan.join("\n"); + assert!( + plan.contains("idx_user_ops_sender_nonce"), + "nonce lookup must seek the sender index:\n{plan}" + ); + assert!( + !plan.contains("TEMP B-TREE"), + "index order must serve ORDER BY:\n{plan}" + ); + } } diff --git a/sequencer/src/storage/migrations/0001_schema.sql b/sequencer/src/storage/migrations/0001_schema.sql index f93e691..cfbacac 100644 --- a/sequencer/src/storage/migrations/0001_schema.sql +++ b/sequencer/src/storage/migrations/0001_schema.sql @@ -231,6 +231,13 @@ CREATE TABLE IF NOT EXISTS user_ops ( FOREIGN KEY(batch_index, frame_in_batch) REFERENCES frames(batch_index, frame_in_batch) ); +-- `GET /nonce` reads a sender's latest op in a valid batch. Rows are never +-- pruned, so without this index every public read would scan all history. +-- It is the chunk commit's costliest index: sender-keyed entries land on +-- scattered leaves, dirtying about one WAL page per distinct sender per chunk. +CREATE INDEX IF NOT EXISTS idx_user_ops_sender_nonce + ON user_ops(sender, nonce); + CREATE TRIGGER IF NOT EXISTS trg_user_ops_target_must_be_tip BEFORE INSERT ON user_ops FOR EACH ROW From 3c78f6a7141dad8be3230a59d16ffaebdd9ad6a4 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Fri, 25 Sep 2026 09:42:03 -0300 Subject: [PATCH 3/7] test: cross-check served nonces against replay oracles in devnet e2e The harness wallet gains served_next_nonce. Restart, stale-batch recovery, Tip cascade, and cold-replica recovery now assert that GET /nonce agrees with the independent replay wallet, including the drop when recovery invalidates soft-confirmed ops. The rebuild round trip pins the documented gap (a sender idle since the rebuild reads 0) and exactness after its first op in the new era. --- tests/e2e/src/cold_replica.rs | 5 +++++ tests/e2e/src/test_cases.rs | 21 +++++++++++++++++++++ tests/harness/src/wallet.rs | 10 ++++++++++ 3 files changed, 36 insertions(+) diff --git a/tests/e2e/src/cold_replica.rs b/tests/e2e/src/cold_replica.rs index 6b82047..cc1e23f 100644 --- a/tests/e2e/src/cold_replica.rs +++ b/tests/e2e/src/cold_replica.rs @@ -157,6 +157,10 @@ async fn run_scenario(runtime: &mut ManagedSequencer) -> Scenari .expect_no_message_for(Duration::from_millis(100)) .await?; let mut alice_l2 = runtime.wallet_l2(TestSigner::from_default(1)?)?; + assert_eq!( + alice_l2.served_next_nonce().await?, + reference.current_user_nonce(alice_address) + ); alice_l2.set_next_nonce(reference.current_user_nonce(alice_address)); alice_l2.transfer(bob_address, U256::from(6_000)).await?; record(&mut reference_ws, &mut reference, &mut history).await?; @@ -239,6 +243,7 @@ async fn run_scenario(runtime: &mut ManagedSequencer) -> Scenari assert_eq!(quote.fee, quote.recommended_fee); let amount = U256::from(6_000); let expected_balance = balance_before - amount - fee_to_linear(quote.fee); + assert_eq!(alice_l2.served_next_nonce().await?, expected_nonce); alice_l2.set_next_nonce(expected_nonce); alice_l2.transfer(bob_address, amount).await?; let resumed = recovered_ws.expect_user_op_from(alice_address).await?; diff --git a/tests/e2e/src/test_cases.rs b/tests/e2e/src/test_cases.rs index d2c8448..fa44043 100644 --- a/tests/e2e/src/test_cases.rs +++ b/tests/e2e/src/test_cases.rs @@ -901,6 +901,7 @@ async fn run_restart_and_replay_test(runtime: &mut ManagedSequencer) -> Scenario // Reading the persisted feed alone does not prove the restarted engine // restored its balances and nonce. Require a fresh execution at nonce 1. let mut resumed_alice = runtime.wallet_l2(alice)?; + assert_eq!(resumed_alice.served_next_nonce().await?, 1); resumed_alice.set_next_nonce(1); resumed_alice.transfer(bob_address, U256::from(1)).await?; let resumed = ws_after_restart.expect_user_op_from(alice_address).await?; @@ -1363,9 +1364,15 @@ async fn run_recovery_after_stale_batches_test( // Step 8: Verify new work succeeds after recovery. let mut alice_l2_fresh = runtime.wallet_l2(alice)?; + assert_eq!( + alice_l2_fresh.served_next_nonce().await?, + 0, + "GET /nonce must fall back with the invalidated transfer" + ); alice_l2_fresh .transfer(bob_address, post_recovery_transfer) .await?; + assert_eq!(alice_l2_fresh.served_next_nonce().await?, 1); replay_after.apply(ws_after.expect_user_op_from(alice_address).await?)?; assert_eq!( @@ -1489,9 +1496,18 @@ async fn run_setup_recovery_round_trip_test( // A continuing nonce exercises the recovered host's state as well as the // independently restored replica and the retained reference history. let mut alice_l2_after = runtime.wallet_l2(alice)?; + // Known gap: the rebuilt database has no ops for the folded prefix, so a + // sender idle since the rebuild reads 0 (application contract, "User + // nonces"). Flip this when setup seeds baseline nonces. + assert_eq!(alice_l2_after.served_next_nonce().await?, 0); alice_l2_after.set_next_nonce(1); let post_transfer = U256::from(70_000_u64); alice_l2_after.transfer(bob_address, post_transfer).await?; + assert_eq!( + alice_l2_after.served_next_nonce().await?, + 2, + "stored nonces are absolute, so one op in the new era makes it exact" + ); let message = resumed_ws.expect_user_op_from(alice_address).await?; rollups_harness::replay::apply_ws_message(&mut restored, message.clone())?; replay.apply(message)?; @@ -1746,6 +1762,11 @@ async fn run_sequencer_outage_danger_zone_tip_cascade_test( 0, "nonce must reset when Tip cascade rolls back the user op", ); + assert_eq!( + runtime.wallet_l2(alice)?.served_next_nonce().await?, + 0, + "GET /nonce must reset with the cascaded Tip", + ); ws_after.expect_no_message_for(NO_WS_MESSAGE_WAIT).await?; diff --git a/tests/harness/src/wallet.rs b/tests/harness/src/wallet.rs index 56ef5cd..01bb3f0 100644 --- a/tests/harness/src/wallet.rs +++ b/tests/harness/src/wallet.rs @@ -254,6 +254,16 @@ impl WalletL2Client { self.next_nonce = nonce; } + /// What `GET /nonce` reports for this signer. Scenarios compare it with + /// an independent replay oracle instead of signing with it. + pub async fn served_next_nonce(&self) -> HarnessResult { + Ok(self + .client + .get_nonce(self.signer.address()) + .await? + .next_nonce) + } + pub async fn transfer(&mut self, to: Address, amount: U256) -> HarnessResult { self.submit_method(Method::Transfer(Transfer { amount, to })) .await From 92904e21fadb66a4999a53571dc85cbc24021822 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Fri, 25 Sep 2026 09:43:18 -0300 Subject: [PATCH 4/7] docs: record the sender-index commit cost and known optimizations A dated record keeps the 2026-09-25 measurement: the sender index raises the mean 64-op chunk commit from 0.27 to 1.0 ms and p99 from 1.1 to 9 ms on the measured machine, and a lane-written per-sender table compares favorably only while the sender population fits in a few pages. The register gains a "Known optimizations" section for headroom with a known mechanism: the sender index and its alternatives, checkpoints running inside the lane's commit, and per-request read connections. Each names its evidence and revisit trigger; the index's schema comment points there. --- .../2026-09-25-sender-index-commit-cost.md | 73 +++++++++++++++++++ docs/review/register.md | 41 +++++++++++ .../src/storage/migrations/0001_schema.sql | 1 + 3 files changed, 115 insertions(+) create mode 100644 docs/review/2026-09-25-sender-index-commit-cost.md diff --git a/docs/review/2026-09-25-sender-index-commit-cost.md b/docs/review/2026-09-25-sender-index-commit-cost.md new file mode 100644 index 0000000..714a4aa --- /dev/null +++ b/docs/review/2026-09-25-sender-index-commit-cost.md @@ -0,0 +1,73 @@ +# Sender-index cost on the chunk commit — 2026-09-25 + +Scope: how much `idx_user_ops_sender_nonce`, which serves `GET /nonce`, adds to +the inclusion lane's chunk commit, and how a lane-written per-sender table +compares. Measured on the storage write path committed in `29fa9fb`. + +Retained for the register entry on the +[sender index](register.md#known-optimizations). Replace it with a +production-like measurement when that entry is revisited; delete it when the +index decision no longer uses these numbers. + +## Environment + +- Apple M5 Max (18 cores, 36 GiB), macOS 26.6.2, APFS on the internal SSD. +- Rust 1.95.0 release build, bundled SQLite 3.53.2 (`libsqlite3-sys` 0.38.1). +- The pragmas `Storage::open` sets in production: WAL and `synchronous=FULL`, + with SQLite's default 1,000-page `wal_autocheckpoint` and 2 MiB page cache. + On macOS, `FULL` calls `fsync` without `F_FULLFSYNC`, so a commit does not + wait for the drive cache. That is cheaper than `fsync` on typical cloud block + storage. + +## Method + +A throwaway `#[ignore]` test in `storage::ingress::tests`, not committed: + +- A fresh database per run, `initialize_open_state`, then 3,000 chunks of 64 + ops (192,000 ops) through `append_executed_user_ops_chunk`, the lane's + production commit path. The timer wraps only that call. +- One open frame for the whole run: no batch closes, dumps, or frame + rotations. Ops carry an empty payload and a 65-byte signature. No concurrent + readers. +- Two sender workloads from a seeded xorshift: every op from a new sender, and + ops drawn from 1,000 senders. +- Three variants: `index` (the committed schema); `none` + (`DROP INDEX idx_user_ops_sender_nonce`); and `projection` (index dropped, + plus a prototype `WITHOUT ROWID` table mapping sender to next nonce, upserted + per op in the same transaction, without the recovery rewind it would need). +- Each variant ran twice, interleaved. Rounds agreed within about 5%. + +## Results + +Per 64-op chunk commit, mean / p50 / p99 over 3,000 commits: + +| Variant | 1,000 senders | New sender per op | +|---|---|---| +| `none` | 0.27 / 0.26 / 1.1 ms | 0.27 / 0.26 / 1.1 ms | +| `index` | 1.0 / 0.53 / 9.1 ms | 1.0 / 0.54 / 9.3 ms | +| `projection` | 0.32 / 0.30 / 1.1 ms | 0.98 / 0.51 / 8.7 ms | + +After 192,000 ops the database file was about 34 MiB without the index, 40 MiB +with it, and 34 or 39 MiB with the projection (1,000 senders or all new). + +## Reading + +- The cost is WAL volume. Sender-keyed entries land on scattered B-tree leaves, + so each distinct sender in a chunk dirties about one more 4 KiB page. The WAL + then reaches the autocheckpoint threshold several times sooner, and the + checkpoint runs inline in the lane's `COMMIT`, which is the p99 column. +- The index pays this at any population size because it grows with total ops. + A per-sender table pays it once the sender population outgrows a few pages: + roughly 9,000 senders for 64-op chunks at about 30 bytes per row. Below that + it stays hot and cheap. +- In absolute terms the index adds about 12 µs per included op on average and + up to about 8 ms at p99 per commit, against the 500 ms ack target. + +## Limits + +- One machine with a cheap `fsync`. Where `fsync` dominates the commit, the + relative overhead is likely smaller; the extra WAL bytes remain. +- No payloads, batch closes, dumps, readers, or application execution. A full + lane turn costs more than the commit alone, so the index is a smaller share + of it. +- The projection figures omit the rewind it would need on recovery. diff --git a/docs/review/register.md b/docs/review/register.md index 49ce13d..6fd70a1 100644 --- a/docs/review/register.md +++ b/docs/review/register.md @@ -109,6 +109,47 @@ exposure in an actual deployment was established by this review. attempt's flush witness. Use the [recovery model](../recovery/README.md) to bound a scenario and assess whether existing component tests suffice. +## Known optimizations + +Performance headroom with a known mechanism, not defects. Each entry names its +measurement, or the lack of one, and the trigger for acting. Checked 2026-09-25 +at `29fa9fb`. + +- **Sender index on the chunk commit.** `idx_user_ops_sender_nonce` lets + `GET /nonce` seek a sender instead of scanning history, but it is the chunk + commit's costliest index. In the + [2026-09-25 measurement](2026-09-25-sender-index-commit-cost.md) it raised + the mean 64-op commit from 0.27 to 1.0 ms and p99 from 1.1 to 9 ms. It stays: + any sender-keyed durable structure pays about the same once the sender + population is large, and a derived index needs no recovery maintenance. If + lane persistence becomes the throughput limit, the alternatives are a + lane-written sender-to-next-nonce table, cheaper only while the sender + population fits in a few pages and needing a rewind on invalidation, and + moving checkpoints off the lane (next entry). Such a table is also where + rebuilt-baseline nonces could be seeded; see + [Track 6](../plans/2026-07-coordination-tracks.md#track-6--dump--application-api-redesign). + Revisit with a measurement on production-like Linux storage, or when lane + throughput is the limit. Evidence: + [`0001_schema.sql`](../../sequencer/src/storage/migrations/0001_schema.sql), + [`storage/ingress.rs`](../../sequencer/src/storage/ingress.rs). +- **Checkpoints run inside the lane's commit.** No connection sets + `wal_autocheckpoint`, so SQLite's default 1,000-page checkpoint runs inline + in the `COMMIT` of whichever writer crosses it, usually the lane. It is the + p99 tail in the measurement above, with or without the sender index. The + candidate is to disable autocheckpoint on the lane's connection and run + `PASSIVE` checkpoints from a background connection; the risk to bound is WAL + growth while readers hold snapshots. Unmeasured beyond that record. Evidence: + [`storage/open.rs`](../../sequencer/src/storage/open.rs). +- **Per-request read connections.** `/fee`, `/nonce`, `/history`, and the + finalized-state routes open a fresh read-only connection per request, so each + pays a file open, a schema parse on its first statement, and cold page and + statement caches. A small pool of long-lived read connections would remove + that cost. Separately, `current_fee_quote` builds a full `WriteHead`, + including two `COUNT(*)` scans of the Tip's user ops, to return two numbers; + that cost is bounded by batch size. Unmeasured. Evidence: + [`ingress/api.rs`](../../sequencer/src/ingress/api.rs), + [`storage/queries.rs`](../../sequencer/src/storage/queries.rs). + ## Verification gaps These are specific behaviors whose coverage remains incomplete, not a mandate diff --git a/sequencer/src/storage/migrations/0001_schema.sql b/sequencer/src/storage/migrations/0001_schema.sql index cfbacac..1299a75 100644 --- a/sequencer/src/storage/migrations/0001_schema.sql +++ b/sequencer/src/storage/migrations/0001_schema.sql @@ -235,6 +235,7 @@ CREATE TABLE IF NOT EXISTS user_ops ( -- pruned, so without this index every public read would scan all history. -- It is the chunk commit's costliest index: sender-keyed entries land on -- scattered leaves, dirtying about one WAL page per distinct sender per chunk. +-- Measured cost and alternatives: docs/review/register.md, "Known optimizations". CREATE INDEX IF NOT EXISTS idx_user_ops_sender_nonce ON user_ops(sender, nonce); From 6b2a233148b803d88e5c09b494ff0d4e0c577624 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Fri, 25 Sep 2026 10:00:48 -0300 Subject: [PATCH 5/7] feat: reject the exhausted user nonce u32::MAX at the shared boundary u32::MAX has no successor under the nonce rule. The wallet accepted an op carrying it when the sender's expected nonce was u32::MAX, then panicked in apply, taking down the lane (or the canonical machine). validate_and_execute_user_op now rejects such an op with InvalidReason::NonceExhausted before app validation, next to the max-fee guard, so the lane and the canonical scheduler agree. Reaching it takes 2^32-1 paid ops from one sender; the rejection is a courtesy, not a live threat. The C header appends APPLICATION_ENGINE_NONCE_EXHAUSTED, carried like the max-fee reason so the vocabulary stays whole; engines never report it, and the host treats it as unsupported if one does. --- AGENTS.md | 2 +- README.md | 1 + .../c-app-engine/include/application-engine.h | 10 ++++--- bindings/c-app-engine/src/lib.rs | 2 +- docs/protocol/application-contract.md | 21 ++++++++------- docs/protocol/scheduler-semantics.md | 5 ++-- examples/app-core/src/application/wallet.rs | 23 ++++++++++++++++ examples/c-wallet-engine/src/lib.rs | 8 +++--- sequencer-core/src/application/mod.rs | 27 ++++++++++++++++--- 9 files changed, 74 insertions(+), 25 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index b97d0f1..500dcdd 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -211,7 +211,7 @@ Paths below are relative to `sequencer/src/`: - API validates the EIP-712 signature and enqueues a `SignedUserOp`. Method payload decoding happens during application execution, not at ingress. - **Deposits are direct-input-only** (L1 → L2) and must not be represented as user ops. -- Rejections (`InvalidNonce`, `InvalidMaxFee`, `InsufficientFeeBalance`) produce no state mutation and are not persisted. These are protocol-level rejection semantics every app must implement: nonces prevent user-op replay, fees prevent spam against the sequencer's DA budget. ("Fee", not "gas" — the fee tracks DA; compute metering, if it ever exists, is a separate future concept.) +- Rejections (`InvalidNonce`, `NonceExhausted`, `InvalidMaxFee`, `InsufficientFeeBalance`) produce no state mutation and are not persisted. These are protocol-level rejection semantics every app must implement: nonces prevent user-op replay, fees prevent spam against the sequencer's DA budget. ("Fee", not "gas" — the fee tracks DA; compute metering, if it ever exists, is a separate future concept.) - **The user-nonce rule is contract, not wallet detail:** accounts start at 0, an op must carry exactly the expected nonce, each included op advances it by one, and direct inputs never touch it. `GET /nonce` derives from persisted user ops under this rule instead of asking the app ([application contract](docs/protocol/application-contract.md#user-nonces)). - Included txs are persisted as frame/batch data in `batches`, `frames`, `user_ops`, `safe_inputs`, and `application_inputs`. Recovery metadata lives in `safe_accepted_batches`; batch lifecycle state (sealed/invalidated) lives on the `batches` row itself as write-once timestamps. - Frame fee is persisted in `frames.fee` and is fixed for the lifetime of that frame. The next frame's fee is currently sampled from `batch_policy_derived.recommended_fee` at rotation; oracle bootstrap writes the price before any Tip can sample it, and `log_slack` applies the 10× margin in log space. This is present behavior, not a reason for the five-block clock policy; hoisting fee to the batch is a later design with its own trade-offs. diff --git a/README.md b/README.md index 0d3a04d..57e4eb2 100644 --- a/README.md +++ b/README.md @@ -239,6 +239,7 @@ Notes: - `sender` is parsed like the `POST /tx` field: `0x` plus 40 hex digits in any case, echoed in EIP-55 casing. A missing or malformed parameter is `400` with code `BAD_REQUEST`. - `next_nonce` is one past the sender's latest included op, or 0 if it has none. It counts soft-confirmed ops: an op acknowledged with `200` before this request is counted. (The `nonce` in a `POST /tx` response or WS message is instead the nonce that op consumed.) - Keep one op in flight per sender: sign `next_nonce`, submit, and increment on `200`. Concurrent submits from one sender are unsupported; they can reach the sequencer out of order and be rejected. After a submit times out, a `next_nonce` above the op's nonce means it was included; an unchanged one is inconclusive, since the op may still be queued, so resubmit the same signed op rather than signing a new one at that nonce. +- `next_nonce` of 4294967295 (`u32::MAX`) means the account is exhausted: that nonce has no successor, so `POST /tx` rejects an op carrying it. - The value is a hint, not a reservation; the application is authoritative. It can go down, because a restart that runs automatic recovery may invalidate soft-confirmed ops. After a `422` bad-nonce rejection, query again rather than incrementing. - After an operator rebuild from a checkpoint (`setup --recovery`), a sender with no op since the rebuild reads 0 even when its nonce in the rebuilt baseline is higher. A `422` bad-nonce rejection names the expected nonce (`bad nonce: expected N, got M`). - Responses carry `Cache-Control: no-store`. `503` with code `UNAVAILABLE` during shutdown. diff --git a/bindings/c-app-engine/include/application-engine.h b/bindings/c-app-engine/include/application-engine.h index 0512578..90e6c40 100644 --- a/bindings/c-app-engine/include/application-engine.h +++ b/bindings/c-app-engine/include/application-engine.h @@ -53,8 +53,8 @@ /// Nonces follow one rule, which the host relies on to serve each sender's next nonce from the /// ops it persisted: a genesis state starts every account at 0, validation accepts only the /// sender's expected nonce, each executed user op advances its sender's nonce by exactly one, and -/// nothing else changes a nonce. An op at UINT32_MAX has no successor and is never executed. See -/// docs/protocol/application-contract.md. +/// nothing else changes a nonce. UINT32_MAX has no successor, so the caller rejects an op carrying +/// it before validation. See docs/protocol/application-contract.md. /// /// Fees are uint16_t exponents with base 129/128, denominated in the fee token's smallest unit. /// The conversion is defined by sequencer-core/src/fee.rs and its build.rs-generated table: @@ -173,6 +173,7 @@ typedef enum ApplicationEngineInvalidReason { APPLICATION_ENGINE_INVALID_NONCE = 0, ///< Nonce or account binding, read `nonce`. APPLICATION_ENGINE_INVALID_MAX_FEE = 1, ///< The caller-owned max fee guard, read `max_fee`. APPLICATION_ENGINE_INSUFFICIENT_FEE_BALANCE = 2, ///< Cannot cover the frame fee, read `fee_balance`. + APPLICATION_ENGINE_NONCE_EXHAUSTED = 3, ///< The caller-owned UINT32_MAX nonce guard, no values. } ApplicationEngineInvalidReason; /// @brief Diagnostics for APPLICATION_ENGINE_INVALID_NONCE. @@ -296,8 +297,9 @@ APPLICATION_ENGINE_API void application_engine_destroy(ApplicationEngine *engine /// @param out_invalid Why the op was refused, written whole and only on INVALID. /// @returns OK, INVALID with diagnostics, IO_ERROR, or INTERNAL_ERROR. /// @details A rejection reports itself through out_invalid and leaves the last error message -/// empty; IO_ERROR and INTERNAL_ERROR carry an error message. The max-fee guard belongs to the -/// caller and is never checked here, so APPLICATION_ENGINE_INVALID_MAX_FEE never comes back. Queued +/// empty; IO_ERROR and INTERNAL_ERROR carry an error message. The max-fee and exhausted-nonce +/// guards belong to the caller and are never checked here, so APPLICATION_ENGINE_INVALID_MAX_FEE +/// and APPLICATION_ENGINE_NONCE_EXHAUSTED never come back. Queued /// outputs are left alone, only an execution touches them. APPLICATION_ENGINE_API ApplicationEngineStatus application_engine_validate_user_op(const ApplicationEngine *engine, const ApplicationEngineEthereumAddress *sender, const ApplicationEngineUserOp *user_op, uint16_t current_fee, diff --git a/bindings/c-app-engine/src/lib.rs b/bindings/c-app-engine/src/lib.rs index 2ec2f35..b6a4b37 100644 --- a/bindings/c-app-engine/src/lib.rs +++ b/bindings/c-app-engine/src/lib.rs @@ -190,7 +190,7 @@ impl Application for EngineApp { available: U256::from_be_bytes(balance.available.bytes), } } - // Max-fee rejection belongs to the shared execution boundary. + // Max-fee and exhausted-nonce rejections belong to the shared execution boundary. other => { return Err(internal(format!( "engine reported unsupported invalid reason {other}" diff --git a/docs/protocol/application-contract.md b/docs/protocol/application-contract.md index fd7b216..4652457 100644 --- a/docs/protocol/application-contract.md +++ b/docs/protocol/application-contract.md @@ -25,10 +25,11 @@ a reference wallet integration. | `progress() -> ApplicationProgress` | Return the count/clock pair from the engine by value. | Execution callers use `validate_and_execute_user_op`, `execute_valid_user_op`, -and `execute_direct_input`. The first enforces `max_fee >= current_fee` before -app validation. App validation checks the nonce and fee balance; a rejection -is not persisted. Trusted replay uses the stored `ValidUserOp`, whose fee and -sender were established at inclusion, without validating a second time. +and `execute_direct_input`. The first enforces `max_fee >= current_fee` and +rejects the exhausted nonce `u32::MAX` before app validation. App validation +checks the nonce and fee balance; a rejection is not persisted. Trusted replay +uses the stored `ValidUserOp`, whose fee and sender were established at +inclusion, without validating a second time. Before an apply hook, the boundary computes the checked expected successor. After `Ok`, it asserts that the engine reports exactly that successor and @@ -73,8 +74,8 @@ rules. These outcomes have different protocol meanings: - A validation rejection changes no state, consumes no nonce or fee, and is - not persisted. `InvalidReason` covers nonce mismatch, insufficient max fee, - and insufficient fee balance. + not persisted. `InvalidReason` covers nonce mismatch, an exhausted nonce, + insufficient max fee, and insufficient fee balance. - A successfully applied input is included even if the business operation fails or is ignored. It advances progress, and an included user op consumes its sender's nonce ([below](#user-nonces)). The wallet also charges the fee @@ -103,8 +104,9 @@ without asking the engine. - Each included user op, business failures included, advances its sender's expected nonce by exactly one. Nothing else changes a nonce: not direct inputs, not other senders' ops. -- Nonces are `u32`. An op carrying `u32::MAX` has no successor and is never - included. +- Nonces are `u32`. `u32::MAX` has no successor, so the shared execution + boundary rejects an op carrying it (`NonceExhausted`) before app + validation; an account that reaches it is exhausted. Under this rule a sender's expected nonce is one past its latest included op. The exception is a sender with no op since the era baseline. A genesis @@ -112,8 +114,7 @@ baseline puts it at 0, but a [rebuilt baseline](../recovery/cockroach.md) carries nonces the rebuilt database has no ops for, so `GET /nonce` reads 0 until that sender's first op. An engine outside the rule (gapped or keyed nonces, a deposit that resets one) makes `GET /nonce` quote nonces the engine -rejects; one that includes an op at `u32::MAX` trips a fail-loud storage -invariant. +rejects. ### 3. The safe-block clock — `last_executed_safe_block` diff --git a/docs/protocol/scheduler-semantics.md b/docs/protocol/scheduler-semantics.md index 8b5cb81..b76a43d 100644 --- a/docs/protocol/scheduler-semantics.md +++ b/docs/protocol/scheduler-semantics.md @@ -105,8 +105,9 @@ For each frame, in order ([`process_batch_payload`](../../sequencer-core/src/sch [`validate_and_execute_user_op`](../../sequencer-core/src/application/mod.rs) with `(frame.fee_price, frame.safe_block)`. A user op is silently skipped (no state change, no output) when its signature is unrecoverable, when app - `validate_user_op` rejects it, or when the protocol `max_fee ≥ fee_price` - guard fails — the fold is a pure deterministic function and emits no + `validate_user_op` rejects it, or when a protocol guard fails + (`max_fee ≥ fee_price`, or a nonce of `u32::MAX`, which has no + successor) — the fold is a pure deterministic function and emits no diagnostics at the library seam. An `AppError` from either execution path is fatal: there is no canonical diff --git a/examples/app-core/src/application/wallet.rs b/examples/app-core/src/application/wallet.rs index 58540b6..71aa2b3 100644 --- a/examples/app-core/src/application/wallet.rs +++ b/examples/app-core/src/application/wallet.rs @@ -444,6 +444,29 @@ mod tests { ); } + #[test] + fn an_exhausted_nonce_is_rejected_instead_of_overflowing() { + use sequencer_core::application::{ExecutionOutcome, validate_and_execute_user_op}; + + let mut app = WalletApp::new(WalletConfig::default()); + let sender = Address::from_slice(&[0x11; 20]); + app.balances.insert(sender, U256::from(10_u64)); + app.nonces.insert(sender, u32::MAX); + let user_op = UserOp { + nonce: u32::MAX, + max_fee: 0, + data: Vec::::new().into(), + }; + + let result = validate_and_execute_user_op(&mut app, sender, &user_op, 0, 0) + .expect("an exhausted nonce is a rejection, not a fault"); + assert_eq!( + result, + ExecutionOutcome::Invalid(InvalidReason::NonceExhausted) + ); + assert_eq!(app.current_user_nonce(sender), u32::MAX); + } + #[test] fn validation_is_repeatable_and_preserves_the_entire_state() { let mut app = WalletApp::default(); diff --git a/examples/c-wallet-engine/src/lib.rs b/examples/c-wallet-engine/src/lib.rs index 05d6b54..1ab6d59 100644 --- a/examples/c-wallet-engine/src/lib.rs +++ b/examples/c-wallet-engine/src/lib.rs @@ -212,11 +212,11 @@ pub unsafe extern "C" fn application_engine_validate_user_op( }, }, }, - // The caller owns the max-fee guard and this entry point never checks it, so the - // app cannot produce this reason. Reporting it would be a lie about which union + // The caller owns these guards and this entry point never checks them, so the app + // cannot produce these reasons. Reporting one would be a lie about which union // member carries the diagnostics. - InvalidReason::InvalidMaxFee { .. } => { - set_error("the app reported a caller-owned max fee rejection"); + InvalidReason::InvalidMaxFee { .. } | InvalidReason::NonceExhausted => { + set_error("the app reported a caller-owned rejection"); return sys::APPLICATION_ENGINE_STATUS_INTERNAL_ERROR; } }; diff --git a/sequencer-core/src/application/mod.rs b/sequencer-core/src/application/mod.rs index 80ac494..b7dea7f 100644 --- a/sequencer-core/src/application/mod.rs +++ b/sequencer-core/src/application/mod.rs @@ -119,6 +119,8 @@ pub enum InvalidReason { max_fee: u16, base_fee: u16, }, + /// The op carried `u32::MAX`, a nonce with no successor. + NonceExhausted, /// Sender cannot pay the frame fee. "Fee" (not "gas"): the current fee /// tracks DA usage; compute metering, if it ever exists, will be a /// separate concept. @@ -137,6 +139,7 @@ impl fmt::Display for InvalidReason { Self::InvalidMaxFee { max_fee, base_fee } => { write!(f, "max fee {max_fee} below base fee {base_fee}") } + Self::NonceExhausted => write!(f, "nonce {} has no successor", u32::MAX), Self::InsufficientFeeBalance { required, available, @@ -159,8 +162,8 @@ pub trait Application: Send + Sized { /// Pure validation predicate over current app state: the op's nonce /// equals the sender's expected nonce (user replay protection), and the /// sender covers the fee. Must not mutate state. - /// [`validate_and_execute_user_op`] enforces the protocol - /// `max_fee >= current_fee` guard before calling here. Rejection leaves + /// [`validate_and_execute_user_op`] enforces the protocol guards + /// (`max_fee >= current_fee`, nonce below `u32::MAX`) before calling here. Rejection leaves /// the app unchanged; `AppError` is fatal and defines no successor. fn validate_user_op( &self, @@ -258,7 +261,7 @@ pub trait CanonicalState { fn canonical_snapshot_bytes(&self) -> Result, AppError>; } -/// Validate and execute a live user op: protocol guard, app validation, execution. +/// Validate and execute a live user op: protocol guards, app validation, execution. /// /// Live inclusion and the canonical scheduler use this boundary. Trusted /// replay uses [`execute_valid_user_op`] with the persisted validation result. @@ -276,6 +279,11 @@ pub fn validate_and_execute_user_op( base_fee: current_fee, })); } + // Protocol invariant: the nonce rule gives `u32::MAX` no successor, so no + // app state can include an op carrying it. + if user_op.nonce == u32::MAX { + return Ok(ExecutionOutcome::Invalid(InvalidReason::NonceExhausted)); + } if let ValidationOutcome::Reject(reason) = app.validate_user_op(sender, user_op, current_fee)? { return Ok(ExecutionOutcome::Invalid(reason)); @@ -510,6 +518,19 @@ mod tests { assert_eq!(app.applied, 0); } + #[test] + fn protocol_exhausted_nonce_guard_precedes_application_validation() { + let mut app = ProgressApp::new(0); + app.fail_validation = true; + let mut exhausted = user_op(); + exhausted.nonce = u32::MAX; + assert_eq!( + validate_and_execute_user_op(&mut app, Address::ZERO, &exhausted, 0, 9).unwrap(), + ExecutionOutcome::Invalid(InvalidReason::NonceExhausted) + ); + assert_eq!(app.applied, 0); + } + #[test] fn execution_error_is_propagated_without_reading_the_failed_instance() { let mut app = ProgressApp::new(7); From f1a810c9674461e364c4b2570ed082134a0eff0f Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Fri, 25 Sep 2026 10:34:46 -0300 Subject: [PATCH 6/7] docs: file GET /nonce read costs and tighten review-raised caveats Adds the read side to the sender-index record: a lookup walks the sender's invalidated index entries above its current nonce (about 0.11 us each, 16 ms at 100,000), bounded by that sender's own rolled-back volume, while opening a read connection per request costs about 0.2 ms, a hundred times the query. The register entries now carry these numbers and the covering-index option. States that the rebuilt-baseline gap also covers a sender whose post-rebuild ops a later recovery invalidated, and why /domain keeps chainId a JSON number: it is exact for every chain browser wallets accept, and a larger ID fails closed at POST /tx. --- README.md | 4 +-- docs/protocol/application-contract.md | 7 ++-- .../2026-09-25-sender-index-commit-cost.md | 35 +++++++++++++++++-- docs/review/register.md | 31 +++++++++------- 4 files changed, 57 insertions(+), 20 deletions(-) diff --git a/README.md b/README.md index 57e4eb2..60140eb 100644 --- a/README.md +++ b/README.md @@ -241,7 +241,7 @@ Notes: - Keep one op in flight per sender: sign `next_nonce`, submit, and increment on `200`. Concurrent submits from one sender are unsupported; they can reach the sequencer out of order and be rejected. After a submit times out, a `next_nonce` above the op's nonce means it was included; an unchanged one is inconclusive, since the op may still be queued, so resubmit the same signed op rather than signing a new one at that nonce. - `next_nonce` of 4294967295 (`u32::MAX`) means the account is exhausted: that nonce has no successor, so `POST /tx` rejects an op carrying it. - The value is a hint, not a reservation; the application is authoritative. It can go down, because a restart that runs automatic recovery may invalidate soft-confirmed ops. After a `422` bad-nonce rejection, query again rather than incrementing. -- After an operator rebuild from a checkpoint (`setup --recovery`), a sender with no op since the rebuild reads 0 even when its nonce in the rebuilt baseline is higher. A `422` bad-nonce rejection names the expected nonce (`bad nonce: expected N, got M`). +- After an operator rebuild from a checkpoint (`setup --recovery`), a sender with no surviving op since the rebuild reads 0 even when its nonce in the rebuilt baseline is higher; this includes a sender whose post-rebuild ops a later recovery invalidated. A `422` bad-nonce rejection names the expected nonce (`bad nonce: expected N, got M`). - Responses carry `Cache-Control: no-store`. `503` with code `UNAVAILABLE` during shutdown. ### `GET /domain` @@ -252,7 +252,7 @@ The EIP-712 domain `POST /tx` verifies signatures against, keyed as `eth_signTyp { "name": "CartesiAppSequencer", "version": "1", "chainId": 31337, "verifyingContract": "0x..." } ``` -Clients pin their own domain and assert that it matches this one, the way a wallet checks `eth_chainId`. Do not sign with a served domain unchecked: `chainId` and `verifyingContract` are what keep a signature from replaying on another deployment, and the name and version are the same everywhere. +Clients pin their own domain and assert that it matches this one, the way a wallet checks `eth_chainId`. `chainId` is a JSON number, exact in JavaScript below 2^53. That covers every chain browser wallets accept (MetaMask refuses IDs above 4503599627370476); a larger ID fails closed, because a signature over a rounded `chainId` recovers a different sender at `POST /tx`. Do not sign with a served domain unchecked: `chainId` and `verifyingContract` are what keep a signature from replaying on another deployment, and the name and version are the same everywhere. ### `GET /ws/subscribe?era_id=&recovery_generation=&next_input=` diff --git a/docs/protocol/application-contract.md b/docs/protocol/application-contract.md index 4652457..676c681 100644 --- a/docs/protocol/application-contract.md +++ b/docs/protocol/application-contract.md @@ -109,10 +109,11 @@ without asking the engine. validation; an account that reaches it is exhausted. Under this rule a sender's expected nonce is one past its latest included op. -The exception is a sender with no op since the era baseline. A genesis -baseline puts it at 0, but a [rebuilt baseline](../recovery/cockroach.md) +The exception is a sender with no surviving op since the era baseline. A +genesis baseline puts it at 0, but a [rebuilt baseline](../recovery/cockroach.md) carries nonces the rebuilt database has no ops for, so `GET /nonce` reads 0 -until that sender's first op. An engine outside the rule (gapped or keyed +until that sender's first op, and again if a later recovery invalidates all of +its post-rebuild ops. An engine outside the rule (gapped or keyed nonces, a deposit that resets one) makes `GET /nonce` quote nonces the engine rejects. diff --git a/docs/review/2026-09-25-sender-index-commit-cost.md b/docs/review/2026-09-25-sender-index-commit-cost.md index 714a4aa..9f03c1f 100644 --- a/docs/review/2026-09-25-sender-index-commit-cost.md +++ b/docs/review/2026-09-25-sender-index-commit-cost.md @@ -1,8 +1,10 @@ -# Sender-index cost on the chunk commit — 2026-09-25 +# Sender-index costs — 2026-09-25 Scope: how much `idx_user_ops_sender_nonce`, which serves `GET /nonce`, adds to the inclusion lane's chunk commit, and how a lane-written per-sender table -compares. Measured on the storage write path committed in `29fa9fb`. +compares; then what a `GET /nonce` read costs, including after recovery leaves +invalidated rows behind. The write side was measured on the storage path +committed in `29fa9fb`, the read side on `6b2a233`. Retained for the register entry on the [sender index](register.md#known-optimizations). Replace it with a @@ -63,6 +65,32 @@ with it, and 34 or 39 MiB with the projection (1,000 senders or all new). - In absolute terms the index adds about 12 µs per included op on average and up to about 8 ms at p99 per commit, against the 500 ms ack target. +## Read cost + +A second throwaway test built a history in which one sender has 10 valid ops, +then N soft-confirmed ops that a cascade invalidated (`insert_invalid_batch`), +then one resubmitted valid op. Its next nonce is 11, and the lookup walks the N +invalidated index entries above that before finding a valid row. A second +sender with 10 valid ops is the baseline. The table gives averages over +repeated calls. "Warm" reuses one read-only connection; "per request" opens a +fresh one each time, as the handler does. + +| Invalidated rows above the sender's nonce | Warm query | Per request | +|---|---|---| +| none (baseline sender) | 1.5 µs | 0.20 ms | +| 100 | 12 µs | 0.21 ms | +| 1,000 | 0.11 ms | 0.33 ms | +| 10,000 | 1.1 ms | 1.5 ms | +| 100,000 | 16 ms | 16 ms | + +- Opening the connection (file open, schema parse, cold caches) is about + 0.2 ms, roughly a hundred times the query itself. +- The walk costs about 0.11 µs per invalidated row. Only rows above the + sender's current nonce are walked, so the cost is the sender's own rolled-back + volume, and it shrinks as the sender resubmits past its old high-water mark. + Accumulating it takes cascades, which come from liveness failures rather than + anything a client can trigger. + ## Limits - One machine with a cheap `fsync`. Where `fsync` dominates the commit, the @@ -71,3 +99,6 @@ with it, and 34 or 39 MiB with the projection (1,000 senders or all new). lane turn costs more than the commit alone, so the index is a smaller share of it. - The projection figures omit the rewind it would need on recovery. +- The read test kept every invalidated row in one batch, so the validity probe + always hit a cached `batches` row. Rows spread over many batches cost a + little more per row. diff --git a/docs/review/register.md b/docs/review/register.md index 6fd70a1..f564715 100644 --- a/docs/review/register.md +++ b/docs/review/register.md @@ -113,7 +113,7 @@ exposure in an actual deployment was established by this review. Performance headroom with a known mechanism, not defects. Each entry names its measurement, or the lack of one, and the trigger for acting. Checked 2026-09-25 -at `29fa9fb`. +at `6b2a233`. - **Sender index on the chunk commit.** `idx_user_ops_sender_nonce` lets `GET /nonce` seek a sender instead of scanning history, but it is the chunk @@ -128,25 +128,30 @@ at `29fa9fb`. moving checkpoints off the lane (next entry). Such a table is also where rebuilt-baseline nonces could be seeded; see [Track 6](../plans/2026-07-coordination-tracks.md#track-6--dump--application-api-redesign). - Revisit with a measurement on production-like Linux storage, or when lane - throughput is the limit. Evidence: + On the read side, the lookup walks the sender's invalidated entries above its + current nonce: about 0.11 µs each, 16 ms at 100,000, bounded by that sender's + own rolled-back volume. A covering `(sender, nonce, batch_index)` index would + skip the table fetch per entry. Revisit with a measurement on production-like + Linux storage, or when lane throughput is the limit. Evidence: [`0001_schema.sql`](../../sequencer/src/storage/migrations/0001_schema.sql), [`storage/ingress.rs`](../../sequencer/src/storage/ingress.rs). - **Checkpoints run inside the lane's commit.** No connection sets - `wal_autocheckpoint`, so SQLite's default 1,000-page checkpoint runs inline - in the `COMMIT` of whichever writer crosses it, usually the lane. It is the - p99 tail in the measurement above, with or without the sender index. The - candidate is to disable autocheckpoint on the lane's connection and run - `PASSIVE` checkpoints from a background connection; the risk to bound is WAL - growth while readers hold snapshots. Unmeasured beyond that record. Evidence: + `wal_autocheckpoint`, so SQLite's default 1,000-page checkpoint runs inline in + the `COMMIT` of whichever writer crosses it, usually the lane. It is the p99 + tail in the measurement above, with or without the sender index. The candidate + is to disable autocheckpoint on the lane's connection and run `PASSIVE` + checkpoints from a background connection; the risk to bound is WAL growth + while readers hold snapshots. Unmeasured beyond that record. Evidence: [`storage/open.rs`](../../sequencer/src/storage/open.rs). - **Per-request read connections.** `/fee`, `/nonce`, `/history`, and the finalized-state routes open a fresh read-only connection per request, so each pays a file open, a schema parse on its first statement, and cold page and - statement caches. A small pool of long-lived read connections would remove - that cost. Separately, `current_fee_quote` builds a full `WriteHead`, - including two `COUNT(*)` scans of the Tip's user ops, to return two numbers; - that cost is bounded by batch size. Unmeasured. Evidence: + statement caches: about 0.2 ms per request against about 2 µs for the nonce + query itself ([measurement](2026-09-25-sender-index-commit-cost.md#read-cost)). + A small pool of long-lived read connections would remove that cost. Separately, + `current_fee_quote` builds a full `WriteHead`, including two `COUNT(*)` scans + of the Tip's user ops, to return two numbers; that cost is bounded by batch + size and unmeasured. Evidence: [`ingress/api.rs`](../../sequencer/src/ingress/api.rs), [`storage/queries.rs`](../../sequencer/src/storage/queries.rs). From 469d3bed70cb76f6d40ed39d0f30c4f0def269e5 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Mon, 28 Sep 2026 08:31:57 -0300 Subject: [PATCH 7/7] fix: keep bad-nonce diagnostics for ops carrying u32::MAX The exhausted-nonce guard ran before app validation, so any op carrying u32::MAX was rejected as "nonce 4294967295 has no successor", even when the sender expected a lower nonce. The guard now runs after the app accepts the op: an op at the wrong nonce gets the app's "bad nonce: expected N, got 4294967295", and NonceExhausted means exactly that the sender's expected nonce is u32::MAX. Both paths reject, and the canonical scheduler ignores rejection reasons, so consensus is unchanged. The threat-model row now states what GET /nonce exposes: a live per-address count of soft-confirmed ops, published before their batches reach L1. It reveals activity timing, not op contents, and moves only after the op is sequenced. --- README.md | 2 +- .../c-app-engine/include/application-engine.h | 2 +- docs/protocol/application-contract.md | 14 ++++--- docs/threat-model/README.md | 2 +- examples/app-core/src/application/wallet.rs | 11 ++++++ sequencer-core/src/application/mod.rs | 39 ++++++++++++------- 6 files changed, 48 insertions(+), 22 deletions(-) diff --git a/README.md b/README.md index 60140eb..acdbaec 100644 --- a/README.md +++ b/README.md @@ -239,7 +239,7 @@ Notes: - `sender` is parsed like the `POST /tx` field: `0x` plus 40 hex digits in any case, echoed in EIP-55 casing. A missing or malformed parameter is `400` with code `BAD_REQUEST`. - `next_nonce` is one past the sender's latest included op, or 0 if it has none. It counts soft-confirmed ops: an op acknowledged with `200` before this request is counted. (The `nonce` in a `POST /tx` response or WS message is instead the nonce that op consumed.) - Keep one op in flight per sender: sign `next_nonce`, submit, and increment on `200`. Concurrent submits from one sender are unsupported; they can reach the sequencer out of order and be rejected. After a submit times out, a `next_nonce` above the op's nonce means it was included; an unchanged one is inconclusive, since the op may still be queued, so resubmit the same signed op rather than signing a new one at that nonce. -- `next_nonce` of 4294967295 (`u32::MAX`) means the account is exhausted: that nonce has no successor, so `POST /tx` rejects an op carrying it. +- `next_nonce` of 4294967295 (`u32::MAX`) means the account is exhausted: that nonce has no successor, so `POST /tx` can include no further op from it. - The value is a hint, not a reservation; the application is authoritative. It can go down, because a restart that runs automatic recovery may invalidate soft-confirmed ops. After a `422` bad-nonce rejection, query again rather than incrementing. - After an operator rebuild from a checkpoint (`setup --recovery`), a sender with no surviving op since the rebuild reads 0 even when its nonce in the rebuilt baseline is higher; this includes a sender whose post-rebuild ops a later recovery invalidated. A `422` bad-nonce rejection names the expected nonce (`bad nonce: expected N, got M`). - Responses carry `Cache-Control: no-store`. `503` with code `UNAVAILABLE` during shutdown. diff --git a/bindings/c-app-engine/include/application-engine.h b/bindings/c-app-engine/include/application-engine.h index 90e6c40..e332ad9 100644 --- a/bindings/c-app-engine/include/application-engine.h +++ b/bindings/c-app-engine/include/application-engine.h @@ -54,7 +54,7 @@ /// ops it persisted: a genesis state starts every account at 0, validation accepts only the /// sender's expected nonce, each executed user op advances its sender's nonce by exactly one, and /// nothing else changes a nonce. UINT32_MAX has no successor, so the caller rejects an op carrying -/// it before validation. See docs/protocol/application-contract.md. +/// it even when validation accepts it. See docs/protocol/application-contract.md. /// /// Fees are uint16_t exponents with base 129/128, denominated in the fee token's smallest unit. /// The conversion is defined by sequencer-core/src/fee.rs and its build.rs-generated table: diff --git a/docs/protocol/application-contract.md b/docs/protocol/application-contract.md index 676c681..f9af109 100644 --- a/docs/protocol/application-contract.md +++ b/docs/protocol/application-contract.md @@ -25,9 +25,10 @@ a reference wallet integration. | `progress() -> ApplicationProgress` | Return the count/clock pair from the engine by value. | Execution callers use `validate_and_execute_user_op`, `execute_valid_user_op`, -and `execute_direct_input`. The first enforces `max_fee >= current_fee` and -rejects the exhausted nonce `u32::MAX` before app validation. App validation -checks the nonce and fee balance; a rejection is not persisted. Trusted replay +and `execute_direct_input`. The first enforces `max_fee >= current_fee` before +app validation and rejects an accepted op at the exhausted nonce `u32::MAX` +after it. App validation checks the nonce and fee balance; a rejection is not +persisted. Trusted replay uses the stored `ValidUserOp`, whose fee and sender were established at inclusion, without validating a second time. @@ -104,9 +105,10 @@ without asking the engine. - Each included user op, business failures included, advances its sender's expected nonce by exactly one. Nothing else changes a nonce: not direct inputs, not other senders' ops. -- Nonces are `u32`. `u32::MAX` has no successor, so the shared execution - boundary rejects an op carrying it (`NonceExhausted`) before app - validation; an account that reaches it is exhausted. +- Nonces are `u32`. `u32::MAX` has no successor, so an account that reaches + it is exhausted: when app validation accepts an op at `u32::MAX`, the shared + execution boundary rejects it (`NonceExhausted`). Any other op carrying + `u32::MAX` fails app validation with the expected nonce. Under this rule a sender's expected nonce is one past its latest included op. The exception is a sender with no surviving op since the era baseline. A diff --git a/docs/threat-model/README.md b/docs/threat-model/README.md index c6fd49b..3fb1bf0 100644 --- a/docs/threat-model/README.md +++ b/docs/threat-model/README.md @@ -25,7 +25,7 @@ What we are protecting: | Batch-submitter private key | Private | Held in operator infra. Not reachable by the network. | | Sequencer's own code | Trusted during normal operation | Tests/review prevent bugs; runtime invariant checks fail loud. Bugs require diagnosis and correction, followed by an operator rebuild if local state cannot be trusted. See "self-trust" below. | | **L1 mempool and block builders** | **Fully adversarial** | May reorder, delay, drop, or selectively include submitted transactions. Private mempools mean "dropped" is indistinguishable from "delayed indefinitely." | -| HTTP clients at `POST /tx`, `GET /fee`, `GET /nonce`, and `GET /domain` | Untrusted | Arbitrary public callers. May submit malformed, malicious, or replay payloads. The `GET` routes are intentional public reads: the open-frame fee, any sender's pending nonce (already public once its batches land on L1), and the signing domain. | +| HTTP clients at `POST /tx`, `GET /fee`, `GET /nonce`, and `GET /domain` | Untrusted | Arbitrary public callers. May submit malformed, malicious, or replay payloads. The `GET` routes are intentional public reads: the open-frame fee, the signing domain, and any sender's soft-confirmed nonce. The nonce is a live per-address count of included ops, published before their batches reach L1 (up to `max_batch_open`, 2 h by default); only the egress feed carries it otherwise. It reveals activity timing, not op contents, and moves only after the op is sequenced, so a watcher cannot front-run the op it counts. | | WebSocket subscribers at `/ws/subscribe` | Internal, but untrusted for data-exposure | Intended for internal indexers. Treat as public for what is exposed. | | Direct-input senders on L1 | Untrusted | Arbitrary L1 accounts calling InputBox. May submit any calldata. | diff --git a/examples/app-core/src/application/wallet.rs b/examples/app-core/src/application/wallet.rs index 71aa2b3..8a176f1 100644 --- a/examples/app-core/src/application/wallet.rs +++ b/examples/app-core/src/application/wallet.rs @@ -465,6 +465,17 @@ mod tests { ExecutionOutcome::Invalid(InvalidReason::NonceExhausted) ); assert_eq!(app.current_user_nonce(sender), u32::MAX); + + app.nonces.insert(sender, 5); + let result = validate_and_execute_user_op(&mut app, sender, &user_op, 0, 0).unwrap(); + assert_eq!( + result, + ExecutionOutcome::Invalid(InvalidReason::InvalidNonce { + expected: 5, + got: u32::MAX + }), + "a sender that is not exhausted gets the expected nonce" + ); } #[test] diff --git a/sequencer-core/src/application/mod.rs b/sequencer-core/src/application/mod.rs index b7dea7f..78a1325 100644 --- a/sequencer-core/src/application/mod.rs +++ b/sequencer-core/src/application/mod.rs @@ -119,7 +119,7 @@ pub enum InvalidReason { max_fee: u16, base_fee: u16, }, - /// The op carried `u32::MAX`, a nonce with no successor. + /// The sender's expected nonce is `u32::MAX`, which has no successor. NonceExhausted, /// Sender cannot pay the frame fee. "Fee" (not "gas"): the current fee /// tracks DA usage; compute metering, if it ever exists, will be a @@ -162,9 +162,10 @@ pub trait Application: Send + Sized { /// Pure validation predicate over current app state: the op's nonce /// equals the sender's expected nonce (user replay protection), and the /// sender covers the fee. Must not mutate state. - /// [`validate_and_execute_user_op`] enforces the protocol guards - /// (`max_fee >= current_fee`, nonce below `u32::MAX`) before calling here. Rejection leaves - /// the app unchanged; `AppError` is fatal and defines no successor. + /// [`validate_and_execute_user_op`] enforces the protocol + /// `max_fee >= current_fee` guard before calling here, and rejects an + /// accepted op at nonce `u32::MAX` after. Rejection leaves the app + /// unchanged; `AppError` is fatal and defines no successor. fn validate_user_op( &self, sender: Address, @@ -279,15 +280,16 @@ pub fn validate_and_execute_user_op( base_fee: current_fee, })); } - // Protocol invariant: the nonce rule gives `u32::MAX` no successor, so no - // app state can include an op carrying it. - if user_op.nonce == u32::MAX { - return Ok(ExecutionOutcome::Invalid(InvalidReason::NonceExhausted)); - } if let ValidationOutcome::Reject(reason) = app.validate_user_op(sender, user_op, current_fee)? { return Ok(ExecutionOutcome::Invalid(reason)); } + // Protocol invariant: the nonce rule gives `u32::MAX` no successor. Checked + // after validation so any other op carrying it gets the app's bad-nonce + // diagnostics; only a sender actually at `u32::MAX` is exhausted. + if user_op.nonce == u32::MAX { + return Ok(ExecutionOutcome::Invalid(InvalidReason::NonceExhausted)); + } let valid = ValidUserOp { sender, @@ -519,14 +521,25 @@ mod tests { } #[test] - fn protocol_exhausted_nonce_guard_precedes_application_validation() { - let mut app = ProgressApp::new(0); - app.fail_validation = true; + fn protocol_exhausted_nonce_guard_follows_application_validation() { let mut exhausted = user_op(); exhausted.nonce = u32::MAX; + + let mut app = ProgressApp::new(0); + app.reject = true; + assert!( + matches!( + validate_and_execute_user_op(&mut app, Address::ZERO, &exhausted, 0, 9).unwrap(), + ExecutionOutcome::Invalid(InvalidReason::InvalidNonce { .. }) + ), + "an app rejection keeps its bad-nonce diagnostics" + ); + + let mut app = ProgressApp::new(0); assert_eq!( validate_and_execute_user_op(&mut app, Address::ZERO, &exhausted, 0, 9).unwrap(), - ExecutionOutcome::Invalid(InvalidReason::NonceExhausted) + ExecutionOutcome::Invalid(InvalidReason::NonceExhausted), + "an op the app accepts at u32::MAX is still never included" ); assert_eq!(app.applied, 0); }