From 9e516e5b2a66eebffb7fc438b396b72098ddb424 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Sat, 5 Sep 2026 15:16:56 -0300 Subject: [PATCH 1/6] fix(node): complete epoch refunds before advancing --- cartesi-rollups/node/README.md | 24 +- cartesi-rollups/node/src/args.rs | 7 - cartesi-rollups/node/src/epoch_manager/mod.rs | 667 ++++++++++-------- .../node/src/epoch_manager/recovery.rs | 382 +++++----- cartesi-rollups/node/src/lib.rs | 7 +- cartesi-rollups/node/src/provider.rs | 40 ++ cartesi-rollups/node/src/storage/advance.rs | 72 +- .../node/src/storage/completion.rs | 243 +++++++ cartesi-rollups/node/src/storage/mod.rs | 4 +- cartesi-rollups/node/src/storage/queries.rs | 45 +- cartesi-rollups/node/src/storage/snapshots.rs | 12 +- .../node/src/storage/sql/discipline.rs | 54 +- .../node/src/storage/sql/schema.sql | 48 +- docs/epoch-lifecycle.md | 82 +-- docs/node-architecture.md | 106 +-- docs/test-harness.md | 19 +- test/e2e/rollups/dave/reader.lua | 18 + test/e2e/rollups/scenarios/multi_sybil.lua | 33 +- 18 files changed, 1179 insertions(+), 684 deletions(-) create mode 100644 cartesi-rollups/node/src/storage/completion.rs diff --git a/cartesi-rollups/node/README.md b/cartesi-rollups/node/README.md index 5931964b5..e75316c2b 100644 --- a/cartesi-rollups/node/README.md +++ b/cartesi-rollups/node/README.md @@ -24,7 +24,7 @@ The dispute engine (formerly the `cartesi-prt-core` crate): `docs/computation-hash.md` first; this is the arcane part. - `hero/` - the honest player: per tick it observes, plans, and dispatches either one dispute action - join, bisect, seal, prove, or - win by timeout - or one bond-freeing cleanup (`gc_planner`). + win by timeout - or one timeout/child cleanup (`gc_planner`). - `tournament/` - the semantic chain interface: `dispute` owns the recursive, event-derived tournament tree; `domain` defines wire-independent values; `observer` performs the narrow pinned point reads; `reader` maintains the @@ -56,11 +56,23 @@ Running the node requires an Ethereum JSON-RPC gateway and a funded wallet. Reads use `--web3-rpc-url`. Raw signed transactions use `--web3-submit-rpc-url`, which defaults to the read endpoint and may instead name a private relay with revert protection. The signer must be exclusive to -one node process because the node owns its nonce sequence. Production submits -at most one mutation per tick - a settlement step, a Hero action, one cleanup, -or bond recovery - through the single serial transaction lane. With the -default `GAS_LIMIT=15_000_000`, a pool may require balance for that full limit -at the transaction's max fee, plus any join bond or other call value. +one node process because the node owns its nonce sequence. Each tick batches +the applicable dispute or cleanup action, settlement step, and all available +bond recoveries at consecutive nonces from the latest mined count. The next +tick rebuilds the batch from chain state without waiting for receipts. + +The node completes epochs in order: it waits for finalized settlement and its +winning bond recoveries before participating in the next epoch. It resumes the +same unfinished epoch after restart. Other participants may advance meanwhile; +the operating timing assumption allows a modest delay while refunds finish. +The completion cursor is bound to one claimant, so changing signer requires a +fresh state directory. A changed node version or schema also requires a fresh +directory under the node's rebuild policy. + +Fund the whole pending batch. With the default `GAS_LIMIT=15_000_000`, a pool +may require each transaction's full gas limit at its max fee, plus its call +value. These requirements accumulate across the batch, and nested tournaments +each require their own join bond. Here are its arguments: diff --git a/cartesi-rollups/node/src/args.rs b/cartesi-rollups/node/src/args.rs index 1d1118173..c5e7192f0 100644 --- a/cartesi-rollups/node/src/args.rs +++ b/cartesi-rollups/node/src/args.rs @@ -243,13 +243,6 @@ impl NodeConfig { Ok(access) } - /// For workers that only read through their own handle (the - /// epoch manager; the dispute Hero opens its own read-write - /// Storage). Fails fast under write pressure instead of stalling. - pub fn storage_read_only(&self) -> Result { - Storage::open_read_only(&self.state_dir) - } - pub async fn read_provider(&self) -> DynProvider { create_rpc_provider(&self.ethereum_gateway, self.chain_id).await } diff --git a/cartesi-rollups/node/src/epoch_manager/mod.rs b/cartesi-rollups/node/src/epoch_manager/mod.rs index 729571253..e9604b08e 100644 --- a/cartesi-rollups/node/src/epoch_manager/mod.rs +++ b/cartesi-rollups/node/src/epoch_manager/mod.rs @@ -5,7 +5,7 @@ mod error; mod recovery; use self::error::Result; -use self::recovery::BondRecovery; +use self::recovery::plan_recovery; use alloy::primitives::{Address, B256, U256}; use alloy::providers::DynProvider; use log::{debug, info, trace}; @@ -31,49 +31,13 @@ pub struct EpochManager { signer_address: Address, sleep_duration: Duration, storage: Storage, - epoch_hero: (Option>, u64), - bond_recovery: BondRecovery, + epoch_hero: Option>, } -enum EpochReaction { - Absent, - Preparing, - Ticked(HeroTick), -} - -impl EpochReaction { - /// Settlement steps run only while no dispute is being contested: - /// before this epoch has local material (an earlier epoch may - /// still be settling) or once the tournament is won. The step, if - /// any, takes the wave's base nonce; the tick's own wave fills - /// strictly above it, so settlement never queues behind dispute - /// work. - fn wants_settlement(&self) -> bool { - match self { - Self::Absent => true, - Self::Preparing => false, - Self::Ticked(tick) => tick.result() == TournamentResult::Won, - } - } - - /// Recovery is maintenance: it runs only when the current epoch - /// has no clock-bearing work and its state was observed without an - /// error. A non-empty settlement or hero wave adds a second fence - /// in the execution loop. - fn allows_recovery(&self) -> bool { - match self { - Self::Absent => true, - Self::Preparing => false, - Self::Ticked(tick) => tick.result() != TournamentResult::Running, - } - } - - fn into_wave(self) -> Vec { - match self { - Self::Ticked(tick) => tick.into_wave(), - Self::Absent | Self::Preparing => Vec::new(), - } - } +struct EpochTick { + epoch: u64, + wave: Vec, + done: bool, } impl EpochManager { @@ -82,101 +46,121 @@ impl EpochManager { transaction_lane: TransactionLane, consensus_address: Address, signer_address: Address, - storage: Storage, + mut storage: Storage, sleep_duration: Duration, - ) -> Self { - Self { + ) -> Result { + storage.pin_epoch_claimant(signer_address)?; + Ok(Self { arena_sender, transaction_lane, consensus: consensus_address, signer_address, sleep_duration, storage, - epoch_hero: (None, 0), - bond_recovery: BondRecovery::new(signer_address), - } + epoch_hero: None, + }) } pub async fn execution_loop(mut self, shutdown: ShutdownSignal, chain: Chain) -> Result<()> { - let dave_consensus = DaveConsensus::new(self.consensus, chain.provider().clone()); - - // A failed iteration is retried, not fatal: every tick is - // re-derived from storage and chain, so transient provider - // errors (an RPC hiccup, a pinned read landing on a block the - // gateway no longer serves) cost one polling interval, never - // the validator. A BlockOutOfRangeError here killed the node - // mid-dispute on 2026-07-10 and its clocks kept running. - // Consensus violations stay fatal: they are asserts, not - // errors. - loop { - let (wave, recovery_allowed) = match self.try_react_epoch(&chain).await { - Ok(reaction) => { - let mut recovery_allowed = reaction.allows_recovery(); - let settlement = if reaction.wants_settlement() { - match self.plan_settlement(&dave_consensus).await { - Ok(step) => { - if step.is_some() { - recovery_allowed = false; - } - step - } - Err(e) => { - recovery_allowed = false; - log::warn!("settlement planning failed, retrying next tick: {e}"); - None - } - } - } else { - None - }; - ( - settlement - .into_iter() - .chain(reaction.into_wave()) - .collect::>(), - recovery_allowed, - ) - } - Err(e) => { - log::warn!("dispute tick failed, retrying next tick: {e}"); - (Vec::new(), false) - } - }; + while !shutdown.is_requested() { + match self.tick(&chain).await { + // Catch up completed historical epochs without a polling sleep. + Ok(true) => continue, + Ok(false) => {} + Err(e) => log::warn!("epoch tick failed, retrying next tick: {e}"), + } + tokio::select! { biased; + _ = shutdown.requested() => break, + _ = tokio::time::sleep(self.sleep_duration) => {} + } + } + Ok(()) + } - if !wave.is_empty() { - // Submit clock-bearing and settlement work before any - // recovery RPC scan can delay it. - if let Err(e) = self.transaction_lane.submit_wave(wave).await { - log::warn!("wave submission failed, retrying next tick: {e}"); - } - } else if recovery_allowed { - match self.latest_epoch_is_finalized(&dave_consensus).await { - Ok(true) => match self.plan_bond_recovery(&chain).await { - Ok(Some(recovery)) => { - if let Err(e) = self.transaction_lane.submit_wave(vec![recovery]).await - { - log::warn!("bond recovery submission failed, retrying later: {e}"); + async fn tick(&mut self, chain: &Chain) -> Result { + let Some(tick) = self.plan_tick(chain).await? else { + return Ok(false); + }; + if tick.done { + assert!( + tick.wave.is_empty(), + "a completed epoch has no pending actions" + ); + // Release the old dispute before advancing the runner's GC boundary. + self.epoch_hero = None; + self.storage.complete_epoch(tick.epoch)?; + info!( + "epoch {} complete: settlement and bonds finalized", + tick.epoch + ); + } else if !tick.wave.is_empty() { + self.transaction_lane + .submit_wave(tick.wave) + .await + .map_err(crate::hero::error::ReactError::from)?; + } + Ok(tick.done) + } + + async fn plan_tick(&mut self, chain: &Chain) -> Result> { + let Some(epoch) = self.storage.unfinished_epoch()? else { + return Ok(None); + }; + let finalized = chain + .finalized_head() + .await + .map_err(crate::hero::error::ReactError::from)?; + // Ingestion is finalized-only. A later sealed epoch proves this one's + // settlement, but it must not outrun the refund observation's head. + let settled = self.storage.last_sealed_epoch()?.is_some_and(|last| { + last.epoch_number > epoch.epoch_number && last.block_created_number <= finalized.number + }); + let mut wave = Vec::new(); + if !settled { + if self.storage.settlement_info(epoch.epoch_number)?.is_some() { + match self.react_dispute(chain, &epoch).await { + Ok(tick) => { + let won = tick.result() == TournamentResult::Won; + wave.extend(tick.into_wave()); + if won { + let consensus = + DaveConsensus::new(self.consensus, chain.provider().clone()); + match self.plan_settlement(&consensus, epoch.epoch_number).await { + Ok(step) => wave.extend(step), + Err(e) => log::warn!( + "settlement planning failed, retrying next tick: {e}" + ), } } - Ok(None) => {} - Err(e) => { - log::warn!("bond recovery planning failed, retrying next tick: {e}"); - } - }, - Ok(false) => { - trace!("defer bond recovery until the latest sealed epoch is finalized"); - } - Err(e) => { - log::warn!("bond recovery epoch fence failed, retrying next tick: {e}"); } + Err(e) => log::warn!("dispute planning failed, retrying next tick: {e}"), } - } - - tokio::select! { biased; - _ = shutdown.requested() => break Ok(()), - _ = tokio::time::sleep(self.sleep_duration) => {} + } else { + debug!( + "wait for machine-runner to prepare epoch {}", + epoch.epoch_number + ); } } + + // Every applicable refund joins the same batch, including while the + // root is running. A failed scan cannot discard already prepared work. + let refunds_complete = + match plan_recovery(chain, &epoch, self.signer_address, finalized).await { + Ok(recovery) => { + wave.extend(recovery.wave); + recovery.complete + } + Err(e) => { + log::warn!("bond recovery planning failed, retrying next tick: {e}"); + false + } + }; + Ok(Some(EpochTick { + epoch: epoch.epoch_number, + wave, + done: settled && refunds_complete, + })) } /// Plans the next staged-settlement step: a sentry claim when @@ -188,20 +172,25 @@ impl EpochManager { /// which is what stops resubmission within a block of inclusion. /// At most one step is planned so a later settlement step never /// queues behind an earlier step from stale state. - pub async fn plan_settlement( + async fn plan_settlement( &mut self, dave_consensus: &DaveConsensus::DaveConsensusInstance< DynProvider, alloy::network::Ethereum, >, + epoch_number: u64, ) -> Result> { - if let Some(step) = self.plan_sentry_claim(dave_consensus).await? { + if let Some(step) = self.plan_sentry_claim(dave_consensus, epoch_number).await? { return Ok(Some(step)); } - if let Some(step) = self.plan_stage_tournament_result(dave_consensus).await? { + if let Some(step) = self + .plan_stage_tournament_result(dave_consensus, epoch_number) + .await? + { return Ok(Some(step)); } - self.plan_accept_tournament_result(dave_consensus).await + self.plan_accept_tournament_result(dave_consensus, epoch_number) + .await } /// A sentry claims the post-epoch state it computed itself - @@ -213,6 +202,7 @@ impl EpochManager { DynProvider, alloy::network::Ethereum, >, + epoch_number: u64, ) -> Result> { let sentry_id = dave_consensus .getSentryId(self.signer_address) @@ -234,6 +224,9 @@ impl EpochManager { .block(alloy::eips::BlockId::latest()) .call() .await?; + if current_sealed_epoch.epochNumber != U256::from(epoch_number) { + return Ok(None); + } let epoch_number = current_sealed_epoch.epochNumber; let has_claimed = dave_consensus @@ -289,6 +282,7 @@ impl EpochManager { DynProvider, alloy::network::Ethereum, >, + epoch_number: u64, ) -> Result> { let can_stage = dave_consensus .canStageTournamentResult() @@ -296,11 +290,12 @@ impl EpochManager { .call() .await?; - // A failed root is a documented terminal state, not a local - // contradiction: the ticked Hero path already logs and idles on - // FailedNoWinner, and this path also runs with no Hero (Absent), - // where crashing would loop on restart. stageTournamentResult's - // TournamentFailedNoWinner revert remains the write-side guard. + if can_stage.epochNumber != U256::from(epoch_number) { + return Ok(None); + } + + // A no-winner result cannot settle. Keep the epoch open for operator + // attention; repeated observation of this state is not a contradiction. if can_stage.isTournamentFailed { log::error!( "dispute tournament for epoch {} finished without a winner; settlement is impossible, notify all users!", @@ -355,6 +350,7 @@ impl EpochManager { DynProvider, alloy::network::Ethereum, >, + epoch_number: u64, ) -> Result> { let can_accept = dave_consensus .canAcceptStagedTournamentResult() @@ -362,6 +358,9 @@ impl EpochManager { .call() .await?; + if can_accept.epochNumber != U256::from(epoch_number) { + return Ok(None); + } if !can_accept.isTournamentResultStaged { trace!("staged tournament result not ready to be accepted"); return Ok(None); @@ -402,123 +401,41 @@ impl EpochManager { Ok(None) } - async fn plan_bond_recovery(&mut self, chain: &Chain) -> Result> { - let epochs = self.storage.sealed_epochs()?; - self.bond_recovery - .plan_due(chain, &epochs) - .await - .map_err(crate::hero::error::ReactError::from) - .map_err(Into::into) - } - - /// Keep maintenance out of the nonce lane while a newly sealed - /// epoch is visible at Latest but not yet in the finalized DB. - async fn latest_epoch_is_finalized( - &mut self, - dave_consensus: &DaveConsensus::DaveConsensusInstance< - DynProvider, - alloy::network::Ethereum, - >, - ) -> Result { - let latest = dave_consensus - .getCurrentSealedEpoch() - .block(alloy::eips::BlockId::latest()) - .call() - .await?; - let finalized = self - .storage - .last_sealed_epoch()? - .map(|epoch| epoch.epoch_number); - Ok(finalized_epoch_matches_latest( - finalized, - latest.epochNumber, - )) - } - - async fn try_react_epoch(&mut self, chain: &Chain) -> Result { - // participate in last sealed epoch tournament - if let Some(last_sealed_epoch) = self.storage.last_sealed_epoch()? { - match self - .storage - .settlement_info(last_sealed_epoch.epoch_number)? - { - Some(_) => { - trace!( - "dispute tournaments for epoch {}", - last_sealed_epoch.epoch_number - ); - return self - .react_dispute(chain, &last_sealed_epoch) - .await - .map(EpochReaction::Ticked); - } - None => { - debug!( - "wait for `machine-runner` to insert settlement values for epoch {}", - last_sealed_epoch.epoch_number - ); - return Ok(EpochReaction::Preparing); - } - } + async fn react_dispute(&mut self, chain: &Chain, epoch: &Epoch) -> Result { + if self.epoch_hero.is_none() { + let storage = Storage::new(self.storage.state_dir())?; + self.epoch_hero = Some(Hero::new( + self.arena_sender.clone(), + chain.clone(), + epoch.root_tournament, + epoch.block_created_number, + storage, + epoch.epoch_number, + )?); } - Ok(EpochReaction::Absent) - } - - async fn react_dispute( - &mut self, - chain: &Chain, - last_sealed_epoch: &Epoch, - ) -> Result { - self.get_latest_hero(last_sealed_epoch, chain)?; let tick = self .epoch_hero - .0 .as_mut() - .expect("hero should be instantiated") + .expect("hero initialized above") .tick() .await?; - match tick.result() { TournamentResult::Running => {} TournamentResult::Won => info!( "local commitment won dispute tournament for epoch {}", - last_sealed_epoch.epoch_number + epoch.epoch_number ), TournamentResult::Lost => log::error!( "local commitment lost dispute tournament for epoch {}", - last_sealed_epoch.epoch_number + epoch.epoch_number ), TournamentResult::FailedNoWinner => log::error!( "dispute tournament for epoch {} finished without a winner", - last_sealed_epoch.epoch_number + epoch.epoch_number ), } - Ok(tick) } - - fn get_latest_hero(&mut self, last_sealed_epoch: &Epoch, chain: &Chain) -> Result<()> { - // either the hero has never been instantiated, or the sealed epoch has advanced - // we need to instantiate new epoch hero with appropriate data - if self.epoch_hero.0.is_none() || self.epoch_hero.1 != last_sealed_epoch.epoch_number { - // The hero reads the closed epoch's working set through - // its own storage handle (one connection per thread). - let storage = Storage::new(self.storage.state_dir())?; - - let hero = Hero::new( - self.arena_sender.clone(), - chain.clone(), - last_sealed_epoch.root_tournament, - last_sealed_epoch.block_created_number, - storage, - last_sealed_epoch.epoch_number, - )?; - - self.epoch_hero = (Some(hero), last_sealed_epoch.epoch_number); - } - - Ok(()) - } } fn to_leaf_proof(proof: StoredLeafProof) -> DaveConsensus::LeafProof { @@ -542,18 +459,24 @@ fn vec_u8_to_bytes_32(hash: Vec) -> B256 { B256::from_slice(&hash) } -fn finalized_epoch_matches_latest(finalized: Option, latest: U256) -> bool { - finalized.is_some_and(|epoch| U256::from(epoch) == latest) -} - #[cfg(test)] mod tests { use super::*; use crate::storage::{ LeafProof, MACHINE_MEMORY_PROOF_SIBLING_COUNT, MachineValidityProof, Proof, }; - use alloy::rpc::types::TransactionRequest; - use alloy::sol_types::SolCall; + use crate::tournament::EthArenaSender; + use alloy::{ + network::EthereumWallet, + primitives::{Bytes, TxKind}, + providers::{Provider, ProviderBuilder}, + rpc::types::{Block, Log}, + signers::local::PrivateKeySigner, + sol_types::SolCall, + transports::mock::Asserter, + }; + use cartesi_prt_contracts::tournament::Tournament; + use std::path::Path; fn proof_leaf(data_byte: u8, sibling_byte: u8) -> LeafProof { LeafProof { @@ -605,74 +528,238 @@ mod tests { ); } - fn tick_wave(labels: &[&str]) -> Vec { - labels - .iter() - .map(|label| (label.to_string(), TransactionRequest::default())) - .collect() + fn setup_epochs() -> tempfile::TempDir { + let dir = tempfile::tempdir().unwrap(); + let conn = rusqlite::Connection::open(dir.path().join("db.sqlite3")).unwrap(); + crate::storage::sql::schema::initialize(&conn).unwrap(); + drop(conn); + let mut storage = Storage::new(dir.path()).unwrap(); + let epochs = [ + Epoch { + epoch_number: 0, + input_index_boundary: 0, + root_tournament: Address::repeat_byte(0x10), + block_created_number: 1, + }, + Epoch { + epoch_number: 1, + input_index_boundary: 0, + root_tournament: Address::repeat_byte(0x20), + block_created_number: 20, + }, + ]; + storage + .insert_consensus_data(20, [].iter(), epochs.iter()) + .unwrap(); + dir + } + + fn manager(path: &Path) -> (EpochManager, Chain, Asserter) { + let asserter = Asserter::new(); + let provider = ProviderBuilder::new() + .connect_mocked_client(asserter.clone()) + .erased(); + let signer = PrivateKeySigner::from_bytes(&B256::repeat_byte(0x11)).unwrap(); + let address = signer.address(); + let lane = TransactionLane::new( + provider.clone(), + provider.clone(), + 31337, + EthereumWallet::from(signer), + ); + let manager = EpochManager::new( + Arc::new(EthArenaSender::new(provider.clone())), + lane, + Address::repeat_byte(0xCC), + address, + Storage::new(path).unwrap(), + Duration::ZERO, + ) + .unwrap(); + (manager, Chain::new(provider, Vec::new()), asserter) + } + + fn push_head(asserter: &Asserter, number: u64) { + let mut block: Block = Block::default(); + block.header.inner.number = number; + block.header.hash = B256::repeat_byte(number as u8); + asserter.push_success(&Some(block)); } - fn ticked(result: TournamentResult, labels: &[&str]) -> EpochReaction { - EpochReaction::Ticked(HeroTick::new(result, tick_wave(labels))) + fn push_call(asserter: &Asserter, value: &C::Return) { + asserter.push_success(&Bytes::from(C::abi_encode_returns(value))); } - #[test] - fn settlement_rides_only_undisputed_phases_and_tick_waves_pass_through() { - assert!(EpochReaction::Absent.wants_settlement()); - assert!(!EpochReaction::Preparing.wants_settlement()); - for (result, wants) in [ - (TournamentResult::Running, false), - (TournamentResult::Won, true), - (TournamentResult::Lost, false), - (TournamentResult::FailedNoWinner, false), - ] { - assert_eq!( - ticked(result, &["heroAction", "eliminateMatchByTimeout"]).wants_settlement(), - wants, - "settlement gating for {result:?}" - ); - } + fn push_bond(asserter: &Asserter, disposition: u8, claimer: Address) { + push_call::( + asserter, + &Tournament::bondRecoveryReturn { + disposition, + claimer, + payment: U256::ZERO, + }, + ); + } - assert!(EpochReaction::Absent.into_wave().is_empty()); - assert!(EpochReaction::Preparing.into_wave().is_empty()); - let labels: Vec = ticked( - TournamentResult::Running, - &["heroAction", "eliminateMatchByTimeout"], - ) - .into_wave() - .into_iter() - .map(|(label, _)| label) - .collect(); + fn push_refund_tick(asserter: &Asserter, finalized: u64, disposition: u8, claimer: Address) { + push_head(asserter, finalized); + asserter.push_success(&Vec::::new()); + push_bond(asserter, disposition, claimer); + } + + #[tokio::test] + async fn rotation_and_restart_finish_old_refunds_before_following_the_next_epoch() { + let dir = setup_epochs(); + let (mut first, chain, rpc) = manager(dir.path()); + let us = first.signer_address; + + // Epoch 1 already exists on the finalized chain, but epoch 0 owns the + // cursor. Its refund does not need any machine snapshots or Hero. + push_refund_tick(&rpc, 30, 2, us); + push_head(&rpc, 31); + push_bond(&rpc, 2, us); + let planned = first.plan_tick(&chain).await.unwrap().unwrap(); + assert_eq!(planned.epoch, 0); + assert!(!planned.done); + assert_eq!(planned.wave.len(), 1); assert_eq!( - labels, - vec!["heroAction", "eliminateMatchByTimeout"], - "the tick's wave order is preserved" + planned.wave[0].1.to, + Some(TxKind::Call(Address::repeat_byte(0x10))) ); + assert!(first.epoch_hero.is_none()); + drop(first); + + // Losing a submission or restarting cannot skip that epoch. + let (mut restarted, chain, rpc) = manager(dir.path()); + push_refund_tick(&rpc, 30, 2, us); + push_head(&rpc, 31); + push_bond(&rpc, 2, us); + let retried = restarted.plan_tick(&chain).await.unwrap().unwrap(); + assert_eq!(retried.epoch, 0); + assert_eq!(retried.wave, planned.wave); + + // Mined but unfinalized recovery suppresses the call, not the epoch. + push_refund_tick(&rpc, 30, 2, us); + push_head(&rpc, 31); + push_bond(&rpc, 3, Address::ZERO); + assert!(!restarted.tick(&chain).await.unwrap()); + assert_eq!( + restarted + .storage + .unfinished_epoch() + .unwrap() + .unwrap() + .epoch_number, + 0 + ); + + // Once the payment is final the real tick advances the durable cursor. + push_refund_tick(&rpc, 32, 3, Address::ZERO); + assert!(restarted.tick(&chain).await.unwrap()); + drop(restarted); + let (mut next, chain, rpc) = manager(dir.path()); + assert_eq!( + next.storage + .unfinished_epoch() + .unwrap() + .unwrap() + .epoch_number, + 1 + ); + push_refund_tick(&rpc, 32, 0, Address::ZERO); + let next_tick = next.plan_tick(&chain).await.unwrap().unwrap(); + assert_eq!(next_tick.epoch, 1); + assert!(!next_tick.done); + assert!(rpc.read_q().is_empty()); } - #[test] - fn recovery_runs_only_in_idle_or_terminal_phases() { - assert!(EpochReaction::Absent.allows_recovery()); - assert!(!EpochReaction::Preparing.allows_recovery()); - for (result, allows) in [ - (TournamentResult::Running, false), - (TournamentResult::Won, true), - (TournamentResult::Lost, true), - (TournamentResult::FailedNoWinner, true), - ] { - assert_eq!( - ticked(result, &[]).allows_recovery(), - allows, - "recovery gating for {result:?}" - ); - } + #[tokio::test] + async fn refund_completion_cannot_outrun_finalized_settlement() { + let dir = setup_epochs(); + let (mut manager, chain, rpc) = manager(dir.path()); + // Ingestion can be ahead of the head sampled by this tick. Even a final + // refund cannot retire the epoch before this observation sees settlement. + push_refund_tick(&rpc, 19, 3, Address::ZERO); + assert!(!manager.tick(&chain).await.unwrap()); + assert_eq!( + manager + .storage + .unfinished_epoch() + .unwrap() + .unwrap() + .epoch_number, + 0 + ); + push_refund_tick(&rpc, 20, 3, Address::ZERO); + assert!(manager.tick(&chain).await.unwrap()); + assert_eq!( + manager + .storage + .unfinished_epoch() + .unwrap() + .unwrap() + .epoch_number, + 1 + ); } - #[test] - fn recovery_waits_for_the_latest_epoch_to_reach_finalized_storage() { - assert!(!finalized_epoch_matches_latest(None, U256::ZERO)); - assert!(finalized_epoch_matches_latest(Some(7), U256::from(7))); - assert!(!finalized_epoch_matches_latest(Some(7), U256::from(8))); - assert!(!finalized_epoch_matches_latest(Some(8), U256::from(7))); + #[tokio::test] + async fn settlement_views_for_a_newer_epoch_do_not_stage_or_accept_it() { + let (dir, mut storage) = crate::storage::sql::test_helper::setup_storage(); + let epochs: Vec<_> = (0..2) + .map(|epoch_number| Epoch { + epoch_number, + input_index_boundary: 0, + root_tournament: Address::repeat_byte(epoch_number as u8 + 1), + block_created_number: 1, + }) + .collect(); + storage + .insert_consensus_data(1, [].iter(), epochs.iter()) + .unwrap(); + storage.roll_epoch().unwrap(); + storage.roll_epoch().unwrap(); + // The newer epoch has valid, fully prepared settlement material. Only + // ownership of epoch zero prevents these otherwise applicable actions. + let newer = storage.settlement_info(1).unwrap().unwrap(); + let (mut manager, chain, rpc) = manager(dir.path()); + let consensus = DaveConsensus::new(manager.consensus, chain.provider().clone()); + push_call::( + &rpc, + &DaveConsensus::canStageTournamentResultReturn { + isFinished: true, + isTournamentFailed: false, + isTournamentResultStaged: false, + epochNumber: U256::from(1), + winnerCommitment: B256::from(newer.computation_hash.data()), + winnerPostEpochMachineStateHash: B256::from(newer.final_state), + }, + ); + assert!( + manager + .plan_stage_tournament_result(&consensus, 0) + .await + .unwrap() + .is_none() + ); + push_call::( + &rpc, + &DaveConsensus::canAcceptStagedTournamentResultReturn { + isTournamentResultStaged: true, + doAllSentriesAgreeWithStagedTournamentResult: true, + isClaimStagingPeriodOver: true, + epochNumber: U256::from(1), + stagedPostEpochMachineStateHash: B256::from(newer.final_state), + stagedPostEpochOutputsMerkleRoot: B256::from(newer.outputs_merkle_root()), + }, + ); + assert!( + manager + .plan_accept_tournament_result(&consensus, 0) + .await + .unwrap() + .is_none() + ); + assert!(rpc.read_q().is_empty()); } } diff --git a/cartesi-rollups/node/src/epoch_manager/recovery.rs b/cartesi-rollups/node/src/epoch_manager/recovery.rs index 04706f957..dd9618372 100644 --- a/cartesi-rollups/node/src/epoch_manager/recovery.rs +++ b/cartesi-rollups/node/src/epoch_manager/recovery.rs @@ -1,8 +1,7 @@ // (c) Cartesi and individual authors (see AUTHORS) // SPDX-License-Identifier: Apache-2.0 (see LICENSE) -//! The bond recovery planner: low-priority maintenance over finalized -//! chain state. +//! Recover one epoch's bonds before its local lifecycle completes. //! //! Candidates never come from attacker-writable input. Epoch roots are //! read from our own storage (written from the trusted DaveConsensus @@ -13,14 +12,12 @@ //! reports the winning claimer from the contract's own classification, //! so "did we join and win" needs no join history at all. //! -//! Retirement uses one coherent finalized snapshot: both the event -//! tree and every bond classification are pinned to its hash. Latest +//! Completion uses one coherent finalized snapshot: the event tree +//! ends at that block and every bond classification uses its hash. Latest //! is consulted only to suppress a transaction already observed as //! mined. It can never retire an epoch, so a reorg cannot turn a //! volatile observation into permanent process state. -use std::collections::BTreeSet; - use alloy::primitives::Address; use anyhow::Result; use log::{info, trace}; @@ -37,123 +34,85 @@ const NO_WINNER: u8 = 1; const RECOVERABLE: u8 = 2; const RECOVERED: u8 = 3; -pub struct BondRecovery { - signer_address: Address, - /// Epochs whose whole tournament tree reached terminal bond - /// dispositions; nothing there can ever need recovery again. - completed_epochs: BTreeSet, - /// Recovery is maintenance, not clock-bearing work. Scan at most - /// once per finalized head after a successful complete attempt. - last_scanned_finalized: Option, +pub struct RecoveryTick { + pub wave: Vec, + pub complete: bool, } -impl BondRecovery { - pub fn new(signer_address: Address) -> Self { - Self { - signer_address, - completed_epochs: BTreeSet::new(), - last_scanned_finalized: None, - } +/// Rebuild every outstanding recovery from chain state. Only finalized +/// classifications may complete an epoch; mined payments suppress retries. +pub async fn plan_recovery( + chain: &Chain, + epoch: &Epoch, + claimant: Address, + finalized: ChainHead, +) -> Result { + if epoch.block_created_number > finalized.number { + return Ok(RecoveryTick { + wave: Vec::new(), + complete: false, + }); } - /// Plan at most one recovery for this finalized-head slot. All - /// sealed epochs participate, so epoch rotation and restart do not - /// strand an older root. - pub async fn plan_due( - &mut self, - chain: &Chain, - epochs: &[Epoch], - ) -> Result> { - let finalized = chain.finalized_head().await?; - if self.last_scanned_finalized == Some(finalized) { - return Ok(None); - } - - let recovery = self.plan_at(chain, epochs, finalized).await?; - self.last_scanned_finalized = Some(finalized); - Ok(recovery) - } - - async fn plan_at( - &mut self, - chain: &Chain, - epochs: &[Epoch], - finalized: ChainHead, - ) -> Result> { - let mut candidates = Vec::new(); - - for epoch in epochs { - if self.completed_epochs.contains(&epoch.epoch_number) { - continue; - } - let tree = tournament_tree( - chain, - epoch.root_tournament, - epoch.block_created_number, - finalized.number, - ) + let tree = tournament_tree( + chain, + epoch.root_tournament, + epoch.block_created_number, + finalized.number, + ) + .await?; + let mut candidates = Vec::new(); + let mut tick = RecoveryTick { + wave: Vec::new(), + complete: true, + }; + for tournament in tree { + let contract = tournament::Tournament::new(tournament, chain.provider()); + let recovery = contract + .bondRecovery() + .block(finalized.block_id()) + .call() .await?; - - let mut all_terminal = true; - for tournament in tree { - let contract = tournament::Tournament::new(tournament, chain.provider()); - let recovery = contract - .bondRecovery() - .block(finalized.block_id()) - .call() - .await?; - match candidate_action(recovery.disposition, recovery.claimer, self.signer_address) - { - CandidateAction::Recover => { - candidates.push(tournament); - all_terminal = false; - } - CandidateAction::Keep => { - all_terminal = false; - } - CandidateAction::Retire => {} - } + match recovery.disposition { + RECOVERABLE if recovery.claimer == claimant => { + candidates.push(tournament); + tick.complete = false; } - if all_terminal { - trace!( - "epoch {} retired: every tournament bond is terminal", - epoch.epoch_number - ); - self.completed_epochs.insert(epoch.epoch_number); + RECOVERABLE | RECOVERED | NO_WINNER => {} + TOURNAMENT_RUNNING => tick.complete = false, + other => { + log::warn!("undefined bond disposition {other}; keeping candidate inert"); + tick.complete = false; } } + } - if candidates.is_empty() { - return Ok(None); - } - - // A mined recovery need not be resubmitted while it waits for - // finality. This observation is deliberately not memoized: a - // reorg merely makes the candidate eligible at the next - // finalized-head slot. - let latest = chain.latest_head().await?; - for tournament in candidates { - let contract = tournament::Tournament::new(tournament, chain.provider()); - let recovery = contract - .bondRecovery() - .block(latest.block_id()) - .call() - .await?; - if recovery.disposition == RECOVERED { - trace!("bond recovery for tournament {tournament} is already mined"); - continue; - } + if candidates.is_empty() { + return Ok(tick); + } - info!("plan bond recovery for tournament {tournament}"); - let request = contract - .tryRecoveringBond() - .gas(gas_limit()) - .into_transaction_request(); - return Ok(Some(("tryRecoveringBond".to_string(), request))); + let latest = chain.latest_head().await?; + for tournament in candidates { + let contract = tournament::Tournament::new(tournament, chain.provider()); + let recovery = contract + .bondRecovery() + .block(latest.block_id()) + .call() + .await?; + if recovery.disposition == RECOVERED { + trace!("bond recovery for tournament {tournament} is already mined"); + continue; } - Ok(None) + info!("plan bond recovery for tournament {tournament}"); + let request = contract + .tryRecoveringBond() + .gas(gas_limit()) + .into_transaction_request(); + tick.wave.push(("tryRecoveringBond".to_string(), request)); } + + Ok(tick) } /// Enumerate one epoch's dispute tree root-down. Every address comes @@ -175,32 +134,6 @@ async fn tournament_tree(chain: &Chain, root: Address, from: u64, to: u64) -> Re Ok(tree) } -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum CandidateAction { - Recover, - Keep, - Retire, -} - -/// One candidate's fate from its on-chain disposition: recover what -/// is ours, keep watching a running tournament, retire everything -/// terminal - recovered (by anyone), locked without a winner, or a -/// bond whose winning claimer is someone else (our commitment lost, -/// or we never joined this branch of the tree). -fn candidate_action(disposition: u8, claimer: Address, us: Address) -> CandidateAction { - match disposition { - RECOVERABLE if claimer == us => CandidateAction::Recover, - RECOVERABLE | RECOVERED | NO_WINNER => CandidateAction::Retire, - TOURNAMENT_RUNNING => CandidateAction::Keep, - other => { - // A trusted tournament cannot produce this; stay inert - // rather than fatal on chain data. - log::warn!("undefined bond disposition {other}; keeping candidate inert"); - CandidateAction::Keep - } - } -} - #[cfg(test)] mod tests { use super::*; @@ -287,40 +220,7 @@ mod tests { assert_eq!(request.1.to, Some(TxKind::Call(tournament))); } - #[test] - fn candidate_fate_follows_the_disposition_arms() { - let us = address(1); - let them = address(2); - assert_eq!( - candidate_action(RECOVERABLE, us, us), - CandidateAction::Recover - ); - assert_eq!( - candidate_action(RECOVERABLE, them, us), - CandidateAction::Retire, - "someone else's recoverable bond is not our work" - ); - assert_eq!( - candidate_action(RECOVERED, Address::ZERO, us), - CandidateAction::Retire - ); - assert_eq!( - candidate_action(NO_WINNER, Address::ZERO, us), - CandidateAction::Retire - ); - assert_eq!( - candidate_action(TOURNAMENT_RUNNING, Address::ZERO, us), - CandidateAction::Keep - ); - assert_eq!( - candidate_action(9, Address::ZERO, us), - CandidateAction::Keep, - "undefined dispositions stay inert, never fatal" - ); - } - - /// The Round-1 finding-4 class: a hand-maintained numeric mirror - /// of a Solidity enum needs a drift guard against its source. + /// The generated bindings expose this Solidity enum as a number. #[test] fn bond_disposition_mirror_matches_the_interface() { let source = std::fs::read_to_string(concat!( @@ -356,21 +256,19 @@ mod tests { let f2 = head(11, 0x11); let latest = head(12, 0x12); let (chain, asserter) = mocked_chain(); - let mut recovery = BondRecovery::new(us); - let epochs = [epoch(7, root, 5)]; + let epoch = epoch(7, root, 5); // At F1 the child does not exist yet and the root is still // running. A latest view could already report the root as // recovered, but it is intentionally never queried here. - asserter.push_success(&Some(block(f1))); asserter.push_success(&Vec::::new()); push_bond(&asserter, TOURNAMENT_RUNNING, Address::ZERO); - assert!(recovery.plan_due(&chain, &epochs).await.unwrap().is_none()); - assert!(!recovery.completed_epochs.contains(&7)); + let tick = plan_recovery(&chain, &epoch, us, f1).await.unwrap(); + assert!(tick.wave.is_empty()); + assert!(!tick.complete); // Once F2 includes the child, the same pinned snapshot sees // the terminal root and our recoverable child together. - asserter.push_success(&Some(block(f2))); asserter.push_success(&vec![child_log(root, child, f2)]); asserter.push_success(&Vec::::new()); push_bond(&asserter, RECOVERED, Address::ZERO); @@ -378,18 +276,15 @@ mod tests { asserter.push_success(&Some(block(latest))); push_bond(&asserter, RECOVERABLE, us); - let request = recovery - .plan_due(&chain, &epochs) - .await - .unwrap() - .expect("the finalized child remains recoverable"); - assert_recovers(&request, child); - assert!(!recovery.completed_epochs.contains(&7)); + let tick = plan_recovery(&chain, &epoch, us, f2).await.unwrap(); + assert_eq!(tick.wave.len(), 1); + assert_recovers(&tick.wave[0], child); + assert!(!tick.complete); assert!(asserter.read_q().is_empty()); } #[tokio::test] - async fn latest_suppression_is_not_retirement_and_recovers_after_a_reorg() { + async fn latest_suppression_is_not_completion_and_recovers_after_a_reorg() { let us = address(1); let root = address(2); let f1 = head(20, 0x20); @@ -397,65 +292,114 @@ mod tests { let h1 = head(22, 0x22); let h2 = head(22, 0x32); let (chain, asserter) = mocked_chain(); - let mut recovery = BondRecovery::new(us); - let epochs = [epoch(8, root, 5)]; + let epoch = epoch(8, root, 5); - asserter.push_success(&Some(block(f1))); asserter.push_success(&Vec::::new()); push_bond(&asserter, RECOVERABLE, us); asserter.push_success(&Some(block(h1))); push_bond(&asserter, RECOVERED, Address::ZERO); - assert!(recovery.plan_due(&chain, &epochs).await.unwrap().is_none()); - assert!(!recovery.completed_epochs.contains(&8)); + let tick = plan_recovery(&chain, &epoch, us, f1).await.unwrap(); + assert!(tick.wave.is_empty()); + assert!(!tick.complete); - // The same finalized slot performs no tree or point-read scan. - asserter.push_success(&Some(block(f1))); - assert!(recovery.plan_due(&chain, &epochs).await.unwrap().is_none()); - - // The unfinalized recovery disappears. A new finalized slot - // recomputes from durable state and makes the bond actionable. - asserter.push_success(&Some(block(f2))); + // Retry immediately when the payment disappears, even while the + // finalized head has not advanced. asserter.push_success(&Vec::::new()); push_bond(&asserter, RECOVERABLE, us); asserter.push_success(&Some(block(h2))); push_bond(&asserter, RECOVERABLE, us); - let request = recovery - .plan_due(&chain, &epochs) - .await - .unwrap() - .expect("latest suppression must not survive a later finalized slot"); - assert_recovers(&request, root); + let tick = plan_recovery(&chain, &epoch, us, f1).await.unwrap(); + assert_eq!(tick.wave.len(), 1); + assert_recovers(&tick.wave[0], root); + assert!(!tick.complete); + + asserter.push_success(&Vec::::new()); + push_bond(&asserter, RECOVERED, Address::ZERO); + let tick = plan_recovery(&chain, &epoch, us, f2).await.unwrap(); + assert!(tick.wave.is_empty()); + assert!(tick.complete); assert!(asserter.read_q().is_empty()); } #[tokio::test] - async fn epoch_rotation_keeps_older_roots_and_plans_only_one_recovery() { + async fn every_owned_bond_is_rebuilt_on_the_same_finalized_head() { let us = address(1); - let old_root = address(2); - let new_root = address(3); + let root = address(2); + let child = address(3); let finalized = head(30, 0x30); let latest = head(31, 0x31); let (chain, asserter) = mocked_chain(); - let mut recovery = BondRecovery::new(us); - let epochs = [epoch(8, old_root, 5), epoch(9, new_root, 25)]; + let epoch = epoch(9, root, 5); + + for _ in 0..2 { + asserter.push_success(&vec![child_log(root, child, finalized)]); + asserter.push_success(&Vec::::new()); + push_bond(&asserter, RECOVERABLE, us); + push_bond(&asserter, RECOVERABLE, us); + asserter.push_success(&Some(block(latest))); + push_bond(&asserter, RECOVERABLE, us); + push_bond(&asserter, RECOVERABLE, us); + + let tick = plan_recovery(&chain, &epoch, us, finalized).await.unwrap(); + assert_eq!(tick.wave.len(), 2); + assert_recovers(&tick.wave[0], root); + assert_recovers(&tick.wave[1], child); + assert!(!tick.complete); + } + assert!(asserter.read_q().is_empty()); + } + + #[tokio::test] + async fn recovered_foreign_and_no_winner_bonds_complete_the_epoch() { + let us = address(1); + let root = address(2); + let child = address(3); + let grandchild = address(4); + let finalized = head(30, 0x30); + let (chain, asserter) = mocked_chain(); + let epoch = epoch(9, root, 5); - asserter.push_success(&Some(block(finalized))); + asserter.push_success(&vec![child_log(root, child, head(10, 0x10))]); + asserter.push_success(&vec![child_log(child, grandchild, head(11, 0x11))]); asserter.push_success(&Vec::::new()); push_bond(&asserter, RECOVERED, Address::ZERO); - asserter.push_success(&Vec::::new()); - push_bond(&asserter, RECOVERABLE, us); - asserter.push_success(&Some(block(latest))); - push_bond(&asserter, RECOVERABLE, us); + push_bond(&asserter, RECOVERABLE, address(5)); + push_bond(&asserter, NO_WINNER, Address::ZERO); + + let tick = plan_recovery(&chain, &epoch, us, finalized).await.unwrap(); + assert!(tick.wave.is_empty()); + assert!(tick.complete); + assert!(asserter.read_q().is_empty()); + } + + #[tokio::test] + async fn running_and_undefined_dispositions_keep_the_epoch_incomplete() { + let (chain, asserter) = mocked_chain(); + let epoch = epoch(9, address(2), 5); + + for disposition in [TOURNAMENT_RUNNING, 9] { + asserter.push_success(&Vec::::new()); + push_bond(&asserter, disposition, Address::ZERO); + let tick = plan_recovery(&chain, &epoch, address(1), head(30, 0x30)) + .await + .unwrap(); + assert!(tick.wave.is_empty()); + assert!(!tick.complete); + } + assert!(asserter.read_q().is_empty()); + } + + #[tokio::test] + async fn epoch_created_after_the_snapshot_waits_without_reading_its_tree() { + let (chain, asserter) = mocked_chain(); + let epoch = epoch(9, address(2), 31); - let request = recovery - .plan_due(&chain, &epochs) + let tick = plan_recovery(&chain, &epoch, address(1), head(30, 0x30)) .await - .unwrap() - .expect("the newer epoch is still scanned after the older one retires"); - assert_recovers(&request, new_root); - assert!(recovery.completed_epochs.contains(&8)); - assert!(!recovery.completed_epochs.contains(&9)); + .unwrap(); + assert!(tick.wave.is_empty()); + assert!(!tick.complete); assert!(asserter.read_q().is_empty()); } } diff --git a/cartesi-rollups/node/src/lib.rs b/cartesi-rollups/node/src/lib.rs index ebb9df2db..726fe1acb 100644 --- a/cartesi-rollups/node/src/lib.rs +++ b/cartesi-rollups/node/src/lib.rs @@ -93,9 +93,8 @@ pub async fn run(config: NodeConfig, shutdown: ShutdownSignal) -> Result<()> { let params = config.clone(); let shutdown = shutdown.clone(); tokio::spawn(async move { - // the epoch manager's own handle only reads; the Hero it - // spawns opens its own writer - let storage = params.storage_read_only()?; + // The manager owns the durable epoch-completion cursor. + let storage = params.storage()?; let read_provider = params.read_provider().await; let transaction_lane = params.transaction_lane(read_provider.clone()).await; let chain = Chain::new( @@ -110,7 +109,7 @@ pub async fn run(config: NodeConfig, shutdown: ShutdownSignal) -> Result<()> { params.signer_address, storage, params.sleep_duration, - ); + )?; epoch_manager.execution_loop(shutdown, chain).await?; Ok(()) }) diff --git a/cartesi-rollups/node/src/provider.rs b/cartesi-rollups/node/src/provider.rs index d56f6a345..62f61b8f1 100644 --- a/cartesi-rollups/node/src/provider.rs +++ b/cartesi-rollups/node/src/provider.rs @@ -576,6 +576,46 @@ mod tests { Ok(()) } + #[tokio::test] + async fn rebuilding_a_wave_fills_a_dropped_prefix_and_resumes_partial_inclusion() -> Result<()> + { + let (_anvil, provider, mut lane, signer) = spawn_lane().await?; + // Exactly one of these transfers fits in each block. + provider.anvil_set_block_gas_limit(21_000).await?; + let initial = lane + .submit_wave(wave(&[("first", 0x11), ("second", 0x22)])) + .await?; + assert_eq!(initial[0].verdict, SendVerdict::Submitted); + assert_eq!(initial[1].verdict, SendVerdict::Submitted); + assert_eq!( + provider.anvil_drop_transaction(initial[0].tx_hash).await?, + Some(initial[0].tx_hash) + ); + provider.anvil_mine(Some(1), None).await?; + assert_eq!( + nonces(&provider, signer).await?.0, + 0, + "the tail cannot fill the missing prefix" + ); + + let retry = lane + .submit_wave(wave(&[("first", 0x11), ("second", 0x22)])) + .await?; + assert_eq!(retry[0].nonce, 0); + assert_eq!(retry[0].verdict, SendVerdict::Submitted); + provider.anvil_mine(Some(1), None).await?; + assert_eq!(nonces(&provider, signer).await?.0, 1); + + // The next observation removes the completed action. No local nonce + // queue or receipt journal is needed to put the remainder at nonce 1. + let remainder = lane.submit_wave(wave(&[("second", 0x22)])).await?; + assert_eq!(remainder[0].nonce, 1); + assert_ne!(remainder[0].verdict, SendVerdict::Failed); + provider.anvil_mine(Some(1), None).await?; + assert_eq!(nonces(&provider, signer).await?, (2, 2)); + Ok(()) + } + /// A process restart is invisible to the pool: there is no lane /// state to lose, so resubmission deduplicates and a changed /// intent waits exactly as it would have without the restart. diff --git a/cartesi-rollups/node/src/storage/advance.rs b/cartesi-rollups/node/src/storage/advance.rs index 00fc9340c..5fac4a1b7 100644 --- a/cartesi-rollups/node/src/storage/advance.rs +++ b/cartesi-rollups/node/src/storage/advance.rs @@ -304,9 +304,21 @@ impl Storage { plan.epoch, plan.boundary_input, ); + self.gc_completed_epochs()?; Ok(plan) } + /// The runner collects released epochs even when no inputs are ready. + /// Keeping directory removal here serializes it with runner publication. + fn gc_completed_epochs(&mut self) -> Result<()> { + if let Some(max_epoch) = self.read(collectable_epoch_in)? { + let orphans = self.write(|tx| gc_old_epochs_in(tx, max_epoch))?; + remove_orphan_dirs(&orphans); + sweep_scratch_dirs_at_or_below(&self.state_dir, max_epoch); + } + Ok(()) + } + /// Opens a batch on a working clone of the newest boundary: the /// machine mutates the clone in place, leaving committed /// boundaries untouched. Restart and tick are the same code path: @@ -571,24 +583,15 @@ impl Storage { machine_validity_proof, }; - let orphans = self.write(|tx| { + self.write(|tx| { assert_eq!( roll_ready_in(tx)?, (previous_epoch_number, recorded), "roll readiness changed before settlement commit" ); insert_snapshot_in(tx, new_epoch_number, 0, &state_hash, &dest_dir)?; - insert_settlement_in(tx, &settlement, previous_epoch_number)?; - if previous_epoch_number >= 1 { - gc_old_epochs_in(tx, previous_epoch_number - 1) - } else { - Ok(Vec::new()) - } + insert_settlement_in(tx, &settlement, previous_epoch_number) })?; - remove_orphan_dirs(&orphans); - if previous_epoch_number >= 1 { - sweep_scratch_dirs_at_or_below(&self.state_dir, previous_epoch_number - 1); - } self.log_disk_breakdown(new_epoch_number); @@ -731,12 +734,17 @@ pub(super) fn insert_settlement_in( Ok(()) } -/// Prunes everything at or below `max_epoch`: boundary rows and the -/// settled epochs' dispute caches. Safe on sling_nodes because -/// DaveConsensus settles epoch N before sealing N + 1, so rows at or -/// below max_epoch belong to finished tournaments. Returns orphaned -/// directories for post-commit removal. -pub(super) fn gc_old_epochs_in(tx: &Transaction, max_epoch: u64) -> Result> { +/// Both cursors are monotonic. Their minimum protects the manager's current +/// dispute material and the runner's newest durable boundary independently. +pub(super) fn collectable_epoch_in(tx: &Transaction) -> Result> { + let next_epoch = super::queries::unfinished_epoch_number_in(tx)?; + let (_, machine_epoch, _, _) = super::snapshots::latest_boundary_in(tx)?; + Ok(next_epoch.min(machine_epoch).checked_sub(1)) +} + +/// Prunes released epochs' boundaries and dispute caches, returning orphaned +/// directories for post-commit removal. `max_epoch` comes from both cursors. +fn gc_old_epochs_in(tx: &Transaction, max_epoch: u64) -> Result> { tx.execute( "DELETE FROM epoch_snapshot_info WHERE epoch_number <= ?1", params![u64_to_i64(max_epoch)], @@ -1271,6 +1279,36 @@ mod tests { ); } + #[test] + fn rolling_ahead_does_not_release_the_managers_unfinished_epoch() { + let (_handle, mut storage) = setup_storage(); + storage.pin_epoch_claimant(Address::ZERO).unwrap(); + let epochs: Vec<_> = (0..3) + .map(|epoch_number| Epoch { + epoch_number, + input_index_boundary: 0, + root_tournament: Address::repeat_byte(epoch_number as u8), + block_created_number: 1, + }) + .collect(); + storage + .insert_consensus_data(1, [].iter(), epochs.iter()) + .unwrap(); + let scratch = storage.epoch_directory(0).unwrap(); + for _ in 0..3 { + storage.roll_epoch().unwrap(); + } + assert_eq!(storage.next_input_id().unwrap().epoch_number, 3); + assert!(storage.snapshot_hash(0, 0).unwrap().is_some()); + assert!(scratch.is_dir()); + + storage.complete_epoch(0).unwrap(); + storage.advance_plan().unwrap(); + assert!(storage.snapshot_hash(0, 0).unwrap().is_none()); + assert!(storage.snapshot_hash(1, 0).unwrap().is_some()); + assert!(!scratch.exists()); + } + #[test] #[should_panic(expected = "refusing to roll open epoch")] fn roll_refuses_an_open_epoch() { diff --git a/cartesi-rollups/node/src/storage/completion.rs b/cartesi-rollups/node/src/storage/completion.rs new file mode 100644 index 000000000..f9ca1a5e6 --- /dev/null +++ b/cartesi-rollups/node/src/storage/completion.rs @@ -0,0 +1,243 @@ +// (c) Cartesi and individual authors (see AUTHORS) +// SPDX-License-Identifier: Apache-2.0 (see LICENSE) + +//! The epoch manager's durable completion cursor. Snapshot and scratch +//! collection stays on the machine runner, after it observes this cursor. + +use super::Storage; +use super::convert::u64_to_i64; +use super::error::{Result, StorageError}; +use super::queries::unfinished_epoch_number_in; +use alloy::primitives::Address; + +impl Storage { + /// Completion is claimant-specific: another signer may still have bonds + /// in epochs this claimant has finished and released for collection. + pub fn pin_epoch_claimant(&mut self, claimant: Address) -> Result<()> { + self.write(|tx| { + let stored: Option> = tx + .query_row( + "SELECT claimant FROM epoch_completion WHERE id = 1", + [], + |row| row.get(0), + ) + .map_err(anyhow::Error::from)?; + if let Some(stored) = stored { + let stored = Address::from_slice(&stored); + if stored != claimant { + return Err(anyhow::anyhow!( + "epoch completion belongs to claimant {stored}, not {claimant}; \ + use a new state directory for a different claimant" + ) + .into()); + } + } else { + tx.execute( + "UPDATE epoch_completion SET claimant = ?1 WHERE id = 1", + [claimant.as_slice()], + ) + .map_err(anyhow::Error::from)?; + } + Ok(()) + }) + } + + /// Releases an epoch after finalized settlement and bond recovery. The + /// caller must stop using its Hero before making the epoch collectible. + pub fn complete_epoch(&mut self, epoch_number: u64) -> Result<()> { + self.write(|tx| { + let expected = unfinished_epoch_number_in(tx)?; + if epoch_number != expected { + return Err(StorageError::InconsistentEpoch { + expected, + provided: epoch_number, + }); + } + tx.execute( + "UPDATE epoch_completion SET next_epoch = next_epoch + 1 + WHERE id = 1 AND next_epoch = ?1", + [u64_to_i64(epoch_number)], + ) + .map_err(anyhow::Error::from)?; + Ok(()) + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::storage::Epoch; + use crate::storage::queries::setup_settlement_storage; + use alloy::hex::ToHexExt; + + fn epochs(count: u64) -> Vec { + (0..count) + .map(|epoch_number| Epoch { + epoch_number, + input_index_boundary: 0, + root_tournament: Address::repeat_byte(epoch_number as u8 + 1), + block_created_number: epoch_number + 1, + }) + .collect() + } + + #[test] + fn unfinished_epoch_survives_restart_and_does_not_follow_latest() { + let (dir, mut storage) = setup_settlement_storage(); + assert!(storage.unfinished_epoch().unwrap().is_none()); + assert!(storage.complete_epoch(0).is_err()); + + let epochs = epochs(3); + storage + .insert_consensus_data(3, [].iter(), epochs.iter()) + .unwrap(); + assert!( + storage.complete_epoch(0).is_err(), + "completion requires a claimant" + ); + storage.pin_epoch_claimant(Address::repeat_byte(7)).unwrap(); + assert_eq!(storage.unfinished_epoch().unwrap().unwrap().epoch_number, 0); + storage.complete_epoch(0).unwrap(); + drop(storage); + + let mut restarted = Storage::new(dir.path()).unwrap(); + restarted + .pin_epoch_claimant(Address::repeat_byte(7)) + .unwrap(); + let mismatch = restarted + .pin_epoch_claimant(Address::repeat_byte(8)) + .unwrap_err(); + assert!(mismatch.to_string().contains("use a new state directory")); + // A rejected signer change leaves the original claimant and cursor intact. + restarted + .pin_epoch_claimant(Address::repeat_byte(7)) + .unwrap(); + assert_eq!( + restarted.unfinished_epoch().unwrap().unwrap().epoch_number, + 1 + ); + assert!(restarted.complete_epoch(0).is_err()); + assert!(restarted.complete_epoch(2).is_err()); + assert_eq!( + restarted.unfinished_epoch().unwrap().unwrap().epoch_number, + 1 + ); + restarted.complete_epoch(1).unwrap(); + restarted.complete_epoch(2).unwrap(); + assert!(restarted.unfinished_epoch().unwrap().is_none()); + assert!(restarted.complete_epoch(3).is_err()); + } + + #[test] + fn idle_runner_prunes_only_completed_epochs_and_keeps_its_newest_boundary() { + let (dir, mut storage) = setup_settlement_storage(); + storage.pin_epoch_claimant(Address::repeat_byte(7)).unwrap(); + let epochs = epochs(6); + storage + .insert_consensus_data(3, [].iter(), epochs.iter().take(3)) + .unwrap(); + + for epoch in 0..=3 { + let boundary = dir.path().join("snapshots").join(epoch.to_string()); + std::fs::create_dir_all(&boundary).unwrap(); + storage + .insert_boundary(epoch, 0, &[epoch as u8 + 1; 32], &boundary) + .unwrap(); + storage.epoch_directory(epoch).unwrap(); + storage + .connection + .execute( + "INSERT INTO sling_nodes VALUES (?1, 0, 0, x'00', x'01')", + [u64_to_i64(epoch)], + ) + .unwrap(); + storage + .connection + .execute( + "INSERT INTO tournament_events_watermark VALUES (?1, 1)", + [epochs[epoch as usize].root_tournament.encode_hex()], + ) + .unwrap(); + storage + .connection + .execute( + "INSERT INTO tournament_events VALUES (?1, 1, 0, x'00')", + [epochs[epoch as usize].root_tournament.encode_hex()], + ) + .unwrap(); + } + + // Both ingestion and execution are ahead of the manager. Neither + // startup cleanup nor an idle runner may discard its epoch zero. + storage.sweep_settled_epoch_scratch().unwrap(); + let plan = storage.advance_plan().unwrap(); + assert!(plan.inputs.is_empty()); + assert!(!plan.sealed); + for epoch in 0..=3 { + assert!(storage.snapshot_dir(epoch, 0).unwrap().is_some()); + assert!(dir.path().join(epoch.to_string()).is_dir()); + } + + storage.complete_epoch(0).unwrap(); + storage.complete_epoch(1).unwrap(); + // Cursor publication itself does not delete the runner's files. + assert!(storage.snapshot_dir(0, 0).unwrap().is_some()); + storage.advance_plan().unwrap(); + for epoch in 0..=3 { + let retained = epoch >= 2; + assert_eq!(storage.snapshot_dir(epoch, 0).unwrap().is_some(), retained); + assert_eq!(dir.path().join(epoch.to_string()).is_dir(), retained); + assert_eq!( + dir.path() + .join("snapshots") + .join(epoch.to_string()) + .is_dir(), + retained + ); + let nodes: i64 = storage + .connection + .query_row( + "SELECT COUNT(*) FROM sling_nodes WHERE epoch = ?1", + [u64_to_i64(epoch)], + |row| row.get(0), + ) + .unwrap(); + assert_eq!(nodes, i64::from(retained)); + for table in ["tournament_events", "tournament_events_watermark"] { + let count: i64 = storage + .connection + .query_row( + &format!("SELECT COUNT(*) FROM {table} WHERE root_tournament = ?1"), + [epochs[epoch as usize].root_tournament.encode_hex()], + |row| row.get(0), + ) + .unwrap(); + assert_eq!(count, i64::from(retained)); + } + } + + storage + .insert_consensus_data(6, [].iter(), epochs.iter().skip(3)) + .unwrap(); + for epoch in 2..=4 { + storage.complete_epoch(epoch).unwrap(); + } + storage.advance_plan().unwrap(); + assert!(storage.snapshot_dir(2, 0).unwrap().is_none()); + assert!(storage.snapshot_dir(3, 0).unwrap().is_some()); + assert_eq!(storage.next_input_id().unwrap().epoch_number, 3); + storage.sweep_settled_epoch_scratch().unwrap(); + assert!(dir.path().join("3").is_dir()); + + // Once the runner publishes another boundary, the previously newest + // completed epoch becomes collectible without another completion. + let boundary = dir.path().join("snapshots/4"); + std::fs::create_dir_all(&boundary).unwrap(); + storage.insert_boundary(4, 0, &[5; 32], &boundary).unwrap(); + storage.advance_plan().unwrap(); + assert!(storage.snapshot_dir(3, 0).unwrap().is_none()); + assert_eq!(storage.next_input_id().unwrap().epoch_number, 4); + assert!(boundary.is_dir()); + } +} diff --git a/cartesi-rollups/node/src/storage/mod.rs b/cartesi-rollups/node/src/storage/mod.rs index a7c1b7273..9d2d24a04 100644 --- a/cartesi-rollups/node/src/storage/mod.rs +++ b/cartesi-rollups/node/src/storage/mod.rs @@ -8,7 +8,8 @@ //! Layout: `open` owns //! the connection lifecycle and the transaction closure helpers; //! `ingest` is the blockchain reader's writer role, `advance` the -//! machine runner's, `dispute` the hero's; `queries` is the +//! machine runner's, `dispute` the hero's, `completion` the epoch manager's; +//! `queries` is the //! role-free read surface; `sql` holds the DDL and its discipline //! tests. Every table belongs to one of four mutation classes - //! append-only log, write-once cell, monotonic watermark, prunable @@ -20,6 +21,7 @@ pub mod rollups_machine; pub use error::StorageError; mod advance; +mod completion; mod convert; mod dispute; mod ingest; diff --git a/cartesi-rollups/node/src/storage/queries.rs b/cartesi-rollups/node/src/storage/queries.rs index 551276f17..378129921 100644 --- a/cartesi-rollups/node/src/storage/queries.rs +++ b/cartesi-rollups/node/src/storage/queries.rs @@ -29,29 +29,19 @@ impl Storage { self.read(epoch_count_in) } - /// Every sealed epoch in order: the bond recovery planner's - /// candidate roots, each written from the trusted consensus - /// stream. - pub fn sealed_epochs(&mut self) -> Result> { + /// The manager's next epoch, once its seal has reached finalized ingestion. + pub fn unfinished_epoch(&mut self) -> Result> { self.read(|tx| { - let mut stmt = tx - .prepare_cached( - r#" - SELECT epoch_number, input_index_boundary, root_tournament, - block_created_number - FROM epochs - ORDER BY epoch_number ASC - "#, - ) - .map_err(anyhow::Error::from)?; - - let rows = stmt - .query_map([], row_to_epoch) - .map_err(anyhow::Error::from)?; - rows.collect::, _>>() - .map_err(anyhow::Error::from)? - .into_iter() - .collect::>>() + let epoch = unfinished_epoch_number_in(tx)?; + tx.query_row( + "SELECT epoch_number, input_index_boundary, root_tournament, + block_created_number FROM epochs WHERE epoch_number = ?1", + [u64_to_i64(epoch)], + row_to_epoch, + ) + .optional() + .map_err(anyhow::Error::from)? + .transpose() }) } @@ -347,6 +337,17 @@ fn settlement_value(epoch_number: u64, field: &str, value: Result) -> T { }) } +pub(super) fn unfinished_epoch_number_in(tx: &Transaction) -> Result { + let epoch: i64 = tx + .query_row( + "SELECT next_epoch FROM epoch_completion WHERE id = 1", + [], + |row| row.get(0), + ) + .map_err(anyhow::Error::from)?; + Ok(i64_to_u64(epoch)) +} + fn row_to_epoch(row: &rusqlite::Row) -> rusqlite::Result> { let tournament_str: String = row.get(2)?; let epoch_number: i64 = row.get(0)?; diff --git a/cartesi-rollups/node/src/storage/snapshots.rs b/cartesi-rollups/node/src/storage/snapshots.rs index ccce952fc..2c8e1fd12 100644 --- a/cartesi-rollups/node/src/storage/snapshots.rs +++ b/cartesi-rollups/node/src/storage/snapshots.rs @@ -666,16 +666,10 @@ pub(super) fn sweep_unreferenced_snapshots_in(tx: &Transaction) -> Result/`), the filesystem sibling of - /// gc_old_epochs_in and safe by the same argument: with the - /// machine at epoch M, epochs at or below M - 2 belong to - /// settled disputes. The roll path sweeps as epochs settle; this - /// entry point is the startup ritual's, catching dirs a crash or - /// an older node version left behind. + /// Startup cleanup uses the same manager and runner cursors as live GC; + /// chain progress alone does not release an unfinished epoch's scratch. pub fn sweep_settled_epoch_scratch(&mut self) -> Result<()> { - let machine_epoch = self.next_input_id()?.epoch_number; - if let Some(max_settled) = machine_epoch.checked_sub(2) { + if let Some(max_settled) = self.read(super::advance::collectable_epoch_in)? { sweep_scratch_dirs_at_or_below(self.state_dir(), max_settled); } Ok(()) diff --git a/cartesi-rollups/node/src/storage/sql/discipline.rs b/cartesi-rollups/node/src/storage/sql/discipline.rs index 523d0e589..67c9b61df 100644 --- a/cartesi-rollups/node/src/storage/sql/discipline.rs +++ b/cartesi-rollups/node/src/storage/sql/discipline.rs @@ -118,6 +118,49 @@ fn latest_processed_only_rises_and_never_disappears() { ); } +#[test] +fn epoch_completion_is_a_dense_permanent_cursor() { + let (_dir, conn) = initialized_conn(); + let update = "UPDATE epoch_completion SET next_epoch = ?1 WHERE id = 1"; + + // Completion cannot invent an epoch which ingestion has not observed. + expect_abort(conn.execute(update, [1]), "one ingested epoch"); + conn.execute("INSERT INTO epochs VALUES (0, 0, '0x00', 0)", []) + .unwrap(); + conn.execute("INSERT INTO epochs VALUES (1, 0, '0x01', 0)", []) + .unwrap(); + expect_abort(conn.execute(update, [1]), "pinned claimant"); + conn.execute("UPDATE epoch_completion SET claimant = zeroblob(20)", []) + .unwrap(); + conn.execute("UPDATE epoch_completion SET claimant = zeroblob(20)", []) + .unwrap(); + expect_abort( + conn.execute("UPDATE epoch_completion SET claimant = NULL", []), + "write-once", + ); + expect_abort( + conn.execute("UPDATE epoch_completion SET claimant = ?1", [[1u8; 20]]), + "write-once", + ); + expect_abort(conn.execute(update, [2]), "one ingested epoch"); + conn.execute(update, [1]).unwrap(); + expect_abort(conn.execute(update, [0]), "one ingested epoch"); + expect_abort(conn.execute(update, [1]), "one ingested epoch"); + conn.execute(update, [2]).unwrap(); + + expect_abort( + conn.execute("DELETE FROM epoch_completion", []), + "permanent singleton", + ); + expect_abort( + conn.execute( + "INSERT OR REPLACE INTO epoch_completion (id, next_epoch) VALUES (1, 0)", + [], + ), + "permanent singleton", + ); +} + // // settlement_info: write-once cell per epoch // @@ -350,7 +393,7 @@ fn tournament_events_watermark_only_rises() { /// The grep-level half of the taxonomy check (the plan accepts it as /// such): across the storage module's Rust sources, the only SQL -/// UPDATEs are the two watermark raises, and the only DELETEs are the +/// UPDATEs advance watermarks and the completion cursor; DELETEs are the /// GC statements. New mutations must either fit an existing class or /// change this test alongside a schema trigger. #[test] @@ -382,9 +425,12 @@ fn mutation_taxonomy_holds_at_source_level() { delete_hits.sort(); assert_eq!( update_hits, - vec![("dispute.rs".to_string(), 1), ("ingest.rs".to_string(), 1)], - "the two watermark upserts (tournament events; latest processed block) \ - are the only UPDATEs in the storage module" + vec![ + ("completion.rs".to_string(), 2), + ("dispute.rs".to_string(), 1), + ("ingest.rs".to_string(), 1) + ], + "only claimant pinning, completion, and the two watermarks write in place" ); assert_eq!( delete_hits, diff --git a/cartesi-rollups/node/src/storage/sql/schema.sql b/cartesi-rollups/node/src/storage/sql/schema.sql index 34446a564..8961c78a2 100644 --- a/cartesi-rollups/node/src/storage/sql/schema.sql +++ b/cartesi-rollups/node/src/storage/sql/schema.sql @@ -55,6 +55,15 @@ CREATE TABLE latest_processed ( INSERT INTO latest_processed (id, block) VALUES (1, 0); +CREATE TABLE epoch_completion ( + id INTEGER NOT NULL PRIMARY KEY CHECK (id = 1), + next_epoch INTEGER NOT NULL + CHECK (typeof(next_epoch) = 'integer' AND next_epoch >= 0), + claimant BLOB + CHECK (claimant IS NULL OR (typeof(claimant) = 'blob' AND length(claimant) = 20)) +) WITHOUT ROWID; +INSERT INTO epoch_completion (id, next_epoch) VALUES (1, 0); + CREATE TABLE template_machine ( id INTEGER PRIMARY KEY CHECK (id = 1), state_hash BLOB NOT NULL @@ -236,6 +245,40 @@ BEGIN SELECT RAISE(ABORT, 'latest_processed is a permanent singleton'); END; +-- The manager finishes epochs in order, only after finalized settlement and +-- bond recovery for its pinned claimant. The cursor may point just beyond the +-- ingested epoch prefix; changing signers requires a different state directory. + +CREATE TRIGGER trg_epoch_completion_dense +BEFORE UPDATE OF next_epoch ON epoch_completion +FOR EACH ROW +WHEN NEW.id != OLD.id OR NEW.next_epoch != OLD.next_epoch + 1 + OR OLD.claimant IS NULL + OR NOT EXISTS (SELECT 1 FROM epochs WHERE epoch_number = OLD.next_epoch) +BEGIN + SELECT RAISE(ABORT, 'epoch_completion must advance by one ingested epoch with a pinned claimant'); +END; + +CREATE TRIGGER trg_epoch_completion_claimant +BEFORE UPDATE OF claimant ON epoch_completion +FOR EACH ROW +WHEN OLD.claimant IS NOT NULL AND NEW.claimant IS NOT OLD.claimant +BEGIN + SELECT RAISE(ABORT, 'epoch_completion claimant is write-once'); +END; + +CREATE TRIGGER trg_epoch_completion_no_insert +BEFORE INSERT ON epoch_completion +BEGIN + SELECT RAISE(ABORT, 'epoch_completion is a permanent singleton'); +END; + +CREATE TRIGGER trg_epoch_completion_no_delete +BEFORE DELETE ON epoch_completion +BEGIN + SELECT RAISE(ABORT, 'epoch_completion is a permanent singleton'); +END; + -- settlement_info: write-once cell per epoch. CREATE TRIGGER trg_settlement_info_no_update @@ -294,9 +337,8 @@ END; -- sling_nodes: append-only write-once-verify (the nondeterminism -- tripwire; message and semantics mirror Storage::insert_quartet_nodes) --- plus settled-epoch prune (gc_old_epochs deletes epochs at least two --- behind the live dispute - DaveConsensus settles epoch N before --- sealing N + 1, so those tournaments are finished). +-- plus completed-epoch prune: the manager's cursor releases the dispute +-- material, and GC always retains the machine runner's newest epoch. CREATE TRIGGER trg_sling_nodes_collision BEFORE INSERT ON sling_nodes diff --git a/docs/epoch-lifecycle.md b/docs/epoch-lifecycle.md index 762635f2d..f606ed441 100644 --- a/docs/epoch-lifecycle.md +++ b/docs/epoch-lifecycle.md @@ -40,6 +40,10 @@ Merkle root (zero at genesis), and the root tournament address. hash OR the claim staging period elapsed - with zero sentries, only the period path exists). Accepting settles the epoch and seals the next one; `EpochSealed` fires here. +5. Locally complete: settlement and every remaining bond payment owed to this + node are resolved in finalized state. Only then does its durable completion + cursor advance to the next epoch. Other participants may already be working + on that epoch while this node finishes its refunds. All four mutating entry points (stage, sentry claim, accept, sentry rotation) are gated by `notForeclosed(appContract)`: a foreclosed @@ -59,15 +63,11 @@ state, not a stranded-value bug. Settlement never touches the tournament's bond path: staging and acceptance move no value, and nothing on the consensus path calls -`tryRecoveringBond`. The decoupling is deliberate - consensus liveness -must not depend on the tournament payment path, and no recipient code -runs inside a settlement transaction. Its cost is an obligation: every -node implementation owns driving bond recovery for each retired -tournament as a permanent background duty, or every retired -tournament's balance - the root's and each inner tournament's - stays -locked with no error reported anywhere. The reference driver walks -unretired sealed epochs and their inner descendants; see the node data -flow below. +`tryRecoveringBond`. No recipient code runs inside a settlement transaction. +The node explicitly recovers its winning bonds from the root and linked inner +tournaments. Its local epoch lifecycle includes those refunds; the contract's +settlement lifecycle remains independent. A foreign claimant's payment and a +no-winner tournament's retained balance do not prevent local completion. ## Node data flow @@ -78,14 +78,14 @@ Three worker threads share one SQLite database (see finalized input/epoch logs Ethereum ----------------------------------> blockchain-reader ^ | - | one serial mutation | inputs, epochs + | consecutive-nonce batch | inputs, epochs | v epoch-manager <-------- settlement data --------- SQLite | ^ +-- Hero <--- tournament logs + pinned views --- Ethereum | \--- Solid events + quartet queries ----> SQLite | - +-- settlement / one GC / recovery + +-- settlement + recoveries + completion cursor machine-runner ---- leaves, snapshots, window quartets ----> SQLite ``` @@ -106,35 +106,35 @@ Three worker threads share one SQLite database (see and the three machine leaf proofs for `iflags_Y`, HTIF tohost, and the first TX-buffer block) together with the next epoch's initial snapshot. The TX block itself is the outputs Merkle root. -- epoch-manager (`cartesi-rollups/node/src/epoch_manager`): each iteration - runs the dispute tick first - for the last sealed epoch, instantiate a - `Hero` with the epoch's inputs, leaves, and snapshot, and let it react - to the tournament - then submits through the one transaction lane it - owns; see - [node architecture](node-architecture.md#mutation-scheduling-and-transaction-submission). - A running dispute tick chooses either the Hero's action or, only when the - Hero has none, one cleanup intent; it never submits both. Settlement runs - only when no dispute is being contested, and bond recovery runs only when no - higher-priority mutation is ready. Thus the production loop submits at most - one mutation per tick through one serial nonce owner. - While machine-runner has not yet written the sealed epoch's - settlement info, the tick reports Preparing and no mutation is submitted. - Settlement plans at most one guarded, idempotent step per - tick: submit a sentry claim when the signer is a sentry (always the - locally computed post-epoch hash, never the staged value - claims - stay an independent check); stage the finished tournament's result - after asserting the on-chain winner matches the local settlement - (commitment root AND post-epoch state); accept the staged result once - every sentry agrees or the staging period elapses. Recovery walks every - unretired sealed epoch, so pending old bonds survive epoch rotation and - restart without a stored queue. Candidate discovery starts from epoch roots - recorded from finalized DaveConsensus events and follows only their - `NewInnerTournament` descendants; it never scans attacker-writable candidates - by submitter. If Latest already exposes the next epoch - while finalized ingestion still ends at the previous one, recovery waits; - the node submits no new maintenance into that observed rotation window. The - lane itself is stateless: every send rebuilds from fresh observation at fresh - market fees, and the mempool arbitrates duplicates and replacements. +- epoch-manager (`cartesi-rollups/node/src/epoch_manager`): follows the durable + `epoch_completion` cursor rather than the latest sealed epoch. Until that + epoch settles, its Hero uses the locally computed settlement material to + choose a dispute action or one cleanup. A won root permits the next + settlement step: claim as a sentry, stage the verified result, or accept it + after agreement or the staging period. These calls target only the cursor's + epoch. Every available refund joins the same batch, even while the root is + running. Settled historical epochs need no Hero or local execution before + the manager can finish their refund obligations and advance. + +Recovery starts from the ingested epoch root and follows that tournament's +`NewInnerTournament` descendants. It reads their bond dispositions at one +finalized hash: our recoverable bonds produce calls; recovered, foreign, and +no-winner dispositions need no further payment; running or unknown dispositions +prevent completion. Latest only suppresses already-mined payments. Restart +resumes the durable cursor, which is bound to the configured claimant. + +The lane rebuilds each batch from current observations and assigns consecutive +nonces from the signer's latest mined count, with fresh market fees. The node +accepts ordinary races, retries, and modest participation delay while finishing +the previous epoch. The account must fund the entire batch's fee envelopes and +call values, including nested tournament bonds. See +[node architecture](node-architecture.md#mutation-scheduling-and-transaction-submission). + +Completion releases the old Hero before advancing the cursor. The machine +runner then collects older snapshots and dispute scratch during its next plan, +even when idle. It collects only epochs below both the completion cursor and +its newest machine epoch. A different claimant or incompatible schema needs a +fresh state directory under the node's rebuild policy. Sentry-claim and settlement calldata are semantic commitments, so their contents come from finalized inputs and stored settlement data. Latest may @@ -185,7 +185,7 @@ remain the computation cache. ## Settlement invariant -`EpochManager::try_settle_epoch` asserts that the tournament winner's +`EpochManager::plan_stage_tournament_result` asserts that the tournament winner's commitment equals the locally computed computation hash. Today a mismatch panics the node (see the debts list in `docs/node-architecture.md`); the intended semantics is "this is a critical incident: either our node is diff --git a/docs/node-architecture.md b/docs/node-architecture.md index d6517fe11..cdebbc522 100644 --- a/docs/node-architecture.md +++ b/docs/node-architecture.md @@ -68,9 +68,9 @@ The storage module follows the sequencer's shape: `open.rs` owns connections (WAL, `foreign_keys=ON`, `synchronous=NORMAL`, busy timeout, a read-only opener) and the `read`/`write` closure helpers (Deferred vs Immediate); writer roles live in per-role files - `ingest.rs` -(blockchain-reader), `advance.rs` (machine-runner), `dispute.rs` -(player) - `snapshots.rs` is the boundary store (every machine -store, load, and clean), and `queries.rs` +(blockchain-reader), `advance.rs` (machine-runner), `dispute.rs` (player), and +`completion.rs` (epoch-manager) - `snapshots.rs` is the boundary store (every +machine store, load, and clean), and `queries.rs` is the role-free read surface. Every public operation is one transaction closure. One node process exclusively owns a state directory; SQLite coordinates its worker threads, not multiple node processes. @@ -116,12 +116,19 @@ between builds that share a package version. It attests which schema created this node-owned cache; manual database mutation remains unsupported rather than continuously audited. +The epoch-completion cursor is bound to one claimant address. A different +configured signer requires a fresh state directory: epochs completed for one +claimant may still hold another claimant's bonds. Incompatible schema or node +versions also require rebuilding a fresh state directory from the chain and +template machine. + Main schema (`storage/sql/schema.sql`): - `node_metadata(node_version, schema_fingerprint)` - immutable cache identity - `epochs(epoch_number, input_index_boundary, root_tournament, block_created_number)` - `inputs(epoch_number, input_index_in_epoch, input)` - `latest_processed(block)` - singleton; last finalized block ingested +- `epoch_completion` - singleton; fixed claimant and next unfinished epoch - `settlement_info(epoch_number, computation_hash, final_state, data block and sibling blobs for iflags_Y, HTIF tohost, and the TX buffer)` - the TX data block is the outputs Merkle root @@ -131,7 +138,7 @@ Main schema (`storage/sql/schema.sql`): - `tournament_events(root_tournament, block_number, log_index, raw_log)` + `tournament_events_watermark` - the dispute reader's persisted finalized prefix: prunable derived store (chain-refetchable, deleted with the - settled epoch); rows are final once written and never outrun the + completed epoch); rows are final once written and never outrun the per-dispute finalized watermark Every table belongs to one of four mutation classes - append-only @@ -143,6 +150,13 @@ Snapshot directories are removed only AFTER the transaction that unreferenced their rows commits: a crash may orphan a directory, never dangle a row. +The manager drops an epoch's Hero before recording completion. The runner +collects released snapshots, dispute rows, and scratch directories while +planning its next batch, including idle polls. It collects only epochs below +both the completion cursor and its newest machine epoch. Neither chain +progress nor manager catch-up can delete the runner's newest durable boundary; +startup scratch cleanup uses the same bound. + The runner captures all three settlement leaves from one final machine root, checks their emulator proof metadata, Keccak openings, nonzero `iflags_Y`, and manual `RX_ACCEPTED` HTIF reason, and verifies that root again when publishing @@ -161,8 +175,8 @@ One schema note to know about: Hero construction materializes nothing: `DisputeSource::on_store` reads the input count, the window-root quartet rows prepaid by the machine runner, and the final boundary hash. Below window granularity, disputes replay the - machine like any nested level; leaf runs are not persisted. `gc_old_epochs` - deletes settled epochs' `sling_nodes` rows, window roots included. + machine like any nested level; leaf runs are not persisted. Completed-epoch + collection deletes released `sling_nodes` rows, window roots included. ## Chain ingestion stance @@ -218,21 +232,36 @@ state; deadline-sensitive responses continue to use Foam. ## Mutation scheduling and transaction submission -The epoch manager directly owns the one non-cloneable transaction lane. Every -tick submits at most one mutation: a Hero action, otherwise one cleanup action; -one settlement step when the dispute is no longer contested; or one recovery -action when all higher-priority work is absent. Recovery is sampled at most -once for each newly observed finalized head. Within one tick, clock-bearing -work is selected before maintenance. +The epoch manager follows the durable cursor for its next unfinished epoch. It advances +only after finalized ingestion proves settlement and one finalized view of the +root and its linked descendants shows no remaining bond payment owed to this +claimant. A recovered bond, a foreign claimant, or a no-winner tournament is +complete; running and unknown dispositions are not. Restart resumes that +cursor. The manager handles settled historical epochs using recovery reads +and calls alone, without constructing a Hero or waiting for local execution. + +For the current epoch, each tick combines the Hero's action or one cleanup, +an applicable settlement step, and every available bond recovery in one batch. +Recovery runs even while the root is contested and retries on the same +finalized head. Latest may suppress an already-mined payment, but only finalized +state can complete the epoch. Settlement planners stay bound to the cursor's +epoch even if another participant has already settled it. + +The node deliberately finishes its refunds before participating in the next +epoch. Other participants can advance the chain meanwhile. Operation accepts +this modest participation delay within the dispute allowance; there is no +promise that recovery never delays another action. The lane is stateless. For every submission it reads the account's mined nonce -at Latest, obtains a fresh EIP-1559 fee estimate, signs one fully specified -request, and hands the raw transaction to the configured submission endpoint. +at Latest, obtains a fresh EIP-1559 fee estimate, and signs the batch at +consecutive nonces from that base. It submits each raw transaction to the +configured endpoint in order; a rejected submission does not discard the tail. It does not wait for a receipt. Already-known transactions, underpriced replacements, and stale nonces are ordinary retry states; every later tick rebuilds intent from fresh observation. The mempool or a separately configured revert-protecting endpoint arbitrates races and duplicates. The signer must be -exclusive to one node instance. +exclusive to one node instance and funded for the whole pending batch's fee +envelopes and call values. Nested tournaments require their own join bonds. ## Known debts @@ -267,63 +296,38 @@ Error handling and observability: 6. Every tournament and settlement request carries the configurable `15_000_000` gas default. A pool may require balance for `gas_limit * max_fee_per_gas + value`, not expected gas use; join value is - therefore additional to the fee envelope. Limiting production to one - request per tick removed cumulative wave funding, but a fee spike can still - reject an otherwise affordable action. Per-verb limits and a calibrated - operating funding floor remain pre-mainnet work. + therefore additional to the fee envelope. A batch needs enough balance for + its cumulative fee envelopes and values, including nested join bonds. + Per-verb limits and a calibrated operating funding floor remain pre-mainnet + work. 7. The lane does not observe receipts or mined revert reasons. Revert protection at the submission endpoint may reject stale or racing transactions before inclusion, but the node neither requires that service nor detects a deterministic self-authored revert. Because reverted state remains unchanged, the same intent may be rebuilt and paid for again each tick. - The lane also does not remember a pending transaction's fees or priority: a + The lane also does not remember a pending transaction's fees: a later, different intent at the same mined nonce may wait until the earlier transaction mines, drops, or becomes replaceable at the fresh market quote. Operation assumes that this happens within the dispute clock budget. Preflight or repeated-intent escalation remains pre-mainnet work. -Scheduling and economic liveness: - -8. Bond recovery can starve across continuous epoch rotation. Recovery is - currently vetoed whenever the current Hero reports a running tournament. - After winning and settling one epoch, an always-participating node can join - after the next sealed epoch is finalized and observed, then remain running - even when its Hero and GC wave is empty. An older winning bond may therefore - never reach the recovery planner. The full E2E battery reproduced this - through `multi_sybil`: the correct claim won the root tournament, but that - tournament retained its balance and the node never logged a recovery plan. - Recovery remains permissionless, so this is a node-automation and economic - liveness defect, not a result-selection failure or a permanent protocol - lock. A fix must guarantee eventual service without allowing maintenance to - delay clock-bearing or settlement work. No current Hero wait state has been - shown to be safe for that purpose: even a tournament awaiting closure still - accepts new joins. Candidate classification must derive from one coherent - finalized view, recovery submission must remain bounded, and pending-nonce - behavior on the shared signer must be explicit. Add a scheduler composition - test spanning recoverable epoch N, settlement and rotation, a running epoch - N+1, and eventual recovery, plus a case where urgent work appears while - maintenance is pending. Keep the `multi_sybil` bond-drain assertion. - Structure: -9. The reader uses async recursion for dynamic tournament discovery, and the +8. The reader uses async recursion for dynamic tournament discovery, and the Hero's dispute loop runs inside the epoch manager task. Local machine and proof preparation can therefore pin a runtime worker. Moving local dispute work to the blocking lane remains open. -10. `EpochManager.epoch_hero: (Option, u64)` - anonymous - tuple state machine; `Hero` construction takes a pile of positional - arguments. -11. Commented-out code blocks kept as reference (the test-scaffolding - `instance.rs` snapshot logic) and disabled/empty tests. -12. No graceful-shutdown story for in-flight work: a mid-epoch machine run +9. Commented-out code blocks kept as reference (the test-scaffolding + `instance.rs` snapshot logic) and disabled/empty tests. +10. No graceful-shutdown story for in-flight work: a mid-epoch machine run or mid-dispute reaction is only interrupted at the next poll. Design assumptions: -13. Finalized-only persistence. The tournament reader additionally acts on a +11. Finalized-only persistence. The tournament reader additionally acts on a disposable number-range tail and point views at one sampled hash. It does not prove the tail belongs to that hash's ancestry; stale work is safe because mutators revalidate it, and the next tick rebuilds the tail. -14. One node instance per state dir; SQLite WAL is the only cross-thread +12. One node instance per state dir; SQLite WAL is the only cross-thread coordination. Shared state-directory operation is unsupported and has no process lock or recovery protocol. diff --git a/docs/test-harness.md b/docs/test-harness.md index b670750bd..29e4a1390 100644 --- a/docs/test-harness.md +++ b/docs/test-harness.md @@ -185,8 +185,9 @@ Scenarios (`test/e2e/rollups/scenarios/`): paths. - `multi_sybil`: the permissionless shape - honest plus three sybils, two matches live at once, two active sybils (one pairing may be sybil-vs-sybil), - one silent sybil - whose match dies by a real on-chain timeout. + one silent sybil whose match dies by a real on-chain timeout. It also + restarts the node around settlement, requires the root bond to drain, and + checks that its `BondRecovered` event precedes the node's next-root join. - `kill_join`: SIGKILL at the hero's join decision (see the marker contract above). - `sealed_leaf_timeout_winner` / `sealed_leaf_timeout_both`: construct @@ -340,12 +341,14 @@ The 2026-08-17 five-lane battery exposed two additional scheduling cases: discovered adversary/adversary roots back to their player coroutines; it must not keep assuming players 1 and 2 form the delegated match. - `multi_sybil` reproducibly selected the correct winner but failed its final - bond-recovery assertion both in the battery and in isolation. Do not remove - the assertion or treat merely lengthening its block loop as a fix: the node's - blanket recovery veto for a running current tournament can starve older bonds - across continuous epoch rotation. The production scheduling debt and - required composition tests are tracked in - [node-architecture.md](node-architecture.md#known-debts). + bond-recovery assertion both in the battery and in isolation. At that + revision, the recovery veto for a running current tournament stranded older + bonds across continuous epoch rotation. The current serial completion + lifecycle is described in + [node-architecture.md](node-architecture.md#mutation-scheduling-and-transaction-submission); + the scenario retains the balance assertion and adds recovery-before-join + ordering across restart. The historical battery result does not validate + the revised implementation. Known blind spots, by layer: diff --git a/test/e2e/rollups/dave/reader.lua b/test/e2e/rollups/dave/reader.lua index c10a5b61d..c3fe6502e 100644 --- a/test/e2e/rollups/dave/reader.lua +++ b/test/e2e/rollups/dave/reader.lua @@ -302,6 +302,24 @@ function Reader:root_tournament_winner(address) return self.inner_reader:root_tournament_winner(address) end +function Reader:read_bond_recovered(tournament_address) + local logs = self.inner_reader:_read_logs( + tournament_address, + "BondRecovered(bytes32,address,uint256,uint256)", + { false, false, false }, + "(uint256,uint256)" + ) + local recovered = {} + for index, log in ipairs(logs) do + recovered[index] = { + meta = log.meta, + commitment = Hash:from_digest_hex(log.emited_topics[2]), + claimer = "0x" .. log.emited_topics[3]:sub(-40), + } + end + return recovered +end + function Reader:commitment_exists(tournament, commitment) local commitments = self.inner_reader:read_commitment_joined(tournament) diff --git a/test/e2e/rollups/scenarios/multi_sybil.lua b/test/e2e/rollups/scenarios/multi_sybil.lua index 7e2a1ab22..94a859420 100644 --- a/test/e2e/rollups/scenarios/multi_sybil.lua +++ b/test/e2e/rollups/scenarios/multi_sybil.lua @@ -122,7 +122,12 @@ print "both active sybils have lost" -- The silent sybil's match resolves by timeout (the honest node's -- win or GC sweep) on the way to settlement. -env.wait_until_epoch(2) +local third_epoch = env.wait_until_epoch(2) + +-- Resume from durable state after settlement. Recovery may already have +-- mined; event ordering below covers both sides of that race. +env.dave_node:kill() +env.dave_node:respawn() -- Pin the real chain deletion that contains the silent commitment. It may -- award the other side or eliminate both, but the silent side cannot win. @@ -148,7 +153,7 @@ assert(winner.commitment == commitment) assert(winner.final == commitment:last()) print "Correct claim won against three sybils!" --- On its idle finalized cadence, the node recovers one bond plus a tenth of +-- Before joining the next root, the node recovers one bond plus a tenth of -- the forfeited sybil residuals. The other nine tenths burn, draining the root -- tournament's balance to zero. Inner tournaments the node won drain -- through the same lane. @@ -164,3 +169,27 @@ assert(recovered, "the node did not recover its bond after settlement") assert(env.dave_node:find_log("plan bond recovery"), "the node's recovery planner left no trace") print "node recovered its bond; forfeited sybil reserves burned" + +local next_join +for _ = 1, 120 do + local joins = env.reader.inner_reader:read_commitment_joined(third_epoch.tournament) + if #joins > 0 then + assert(#joins == 1, "the node joined the next root more than once") + next_join = joins[1] + break + end + env.fast_forward(1) +end +assert(next_join, "the node did not join the next root after recovery and restart") + +local recoveries = env.reader:read_bond_recovered(second_epoch.tournament) +assert(#recoveries == 1, "the settled root must emit exactly one bond recovery") +local recovery = recoveries[1] +assert(recovery.commitment == commitment.root_hash, "recovery paid the wrong commitment") +assert(recovery.claimer:lower() == env.dave_node.wallet_address:lower(), + "recovery did not pay the node") +assert(recovery.meta.block_number < next_join.meta.block_number + or (recovery.meta.block_number == next_join.meta.block_number + and recovery.meta.log_index < next_join.meta.log_index), + "the node joined the next root before recovering its settled root bond") +print "root bond recovery preceded the next join across restart" From e7987f333cb4ce5e3fc484dbaa6ccd31010852ad Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Sat, 5 Sep 2026 15:24:45 -0300 Subject: [PATCH 2/6] refactor(node): trim unused ruler abstractions --- .../node/src/engine/machine_stf.rs | 6 +- cartesi-rollups/node/src/engine/mod.rs | 21 +- cartesi-rollups/node/src/engine/ruler.rs | 20 +- cartesi-rollups/node/src/engine/spec.rs | 44 +-- cartesi-rollups/node/src/engine/stf.rs | 254 +---------------- cartesi-rollups/node/src/engine/toy.rs | 269 ++++++++++++++++++ cartesi-rollups/node/src/hero/action.rs | 3 +- cartesi-rollups/node/src/hero/actor.rs | 14 +- cartesi-rollups/node/src/hero/context.rs | 10 +- docs/computation-hash.md | 8 + 10 files changed, 329 insertions(+), 320 deletions(-) create mode 100644 cartesi-rollups/node/src/engine/toy.rs diff --git a/cartesi-rollups/node/src/engine/machine_stf.rs b/cartesi-rollups/node/src/engine/machine_stf.rs index c4fbb9929..7ce5b7e85 100644 --- a/cartesi-rollups/node/src/engine/machine_stf.rs +++ b/cartesi-rollups/node/src/engine/machine_stf.rs @@ -195,7 +195,7 @@ impl MachineStf { } fn terminal_fixed(&mut self) -> Result { - if self.halted()? || self.mcycle_overflow()? { + if self.machine.iflags_h()? || self.mcycle_overflow()? { return Ok(true); } Ok(matches!( @@ -235,10 +235,6 @@ impl Stf for MachineStf { Ok(self.machine.root_hash()?.into()) } - fn halted(&mut self) -> Result { - Ok(self.machine.iflags_h()?) - } - fn yielded(&mut self) -> Result { Ok(self.manual_yield_reason()? == Some(RX_ACCEPTED)) } diff --git a/cartesi-rollups/node/src/engine/mod.rs b/cartesi-rollups/node/src/engine/mod.rs index c427c4611..a50e8036e 100644 --- a/cartesi-rollups/node/src/engine/mod.rs +++ b/cartesi-rollups/node/src/engine/mod.rs @@ -12,23 +12,22 @@ //! docs/computation-hash.md. //! //! Layering, innermost first: -//! - [`stf::Stf`]: machine verbs (ustep, ureset, feed, revert). Two -//! implementations: the toy (here, for spec tests) and the Cartesi -//! machine. +//! - [`stf::Stf`]: the machine operations the ruler needs. Production +//! uses the Cartesi machine; unit tests use a small scripted machine. //! - [`ruler::Ruler`]: the geometry engine. Owns every meta-cycle //! convention (window boundaries, fused feed transition, big-cycle //! closing ureset, fixed-point padding). Written once, exercised by //! the toy, reused by the production machine. -//! - [`cache::NodeCache`] and [`cache::get_or_compute`]: the quartet -//! cache with its amortizing fanout. +//! - [`cache::get_or_compute`]: quartet computation and fanout over +//! the node's storage. //! - [`dispute::DisputeSource`]: the hero-facing face. Tournament //! coordinates map onto quartets ([`dispute::LevelCoords`]), level 0 //! is served from the persisted regime-1 material (window-root rows //! plus lazy interior folds), and proofs are sibling descents. //! -//! The spec tests in `spec.rs` compare all of this against an -//! independent brute-force oracle; they are the executable form of the -//! leaf-convention specification. +//! The spec tests compare stepping and sampling against a literal +//! leaf sequence. Cache and proof tests also use trees built from those +//! runs; real-machine differentials live in `tests/engine_machine.rs`. pub mod cache; pub mod config; @@ -41,10 +40,12 @@ pub mod structure; #[cfg(test)] pub(crate) mod spec; +#[cfg(test)] +pub(crate) mod toy; pub use config::EngineConfig; pub use dispute::{DisputeSource, LevelCoords, fold_runs}; pub use machine_stf::{MachineStf, Positioner}; -pub use ruler::{Ruler, RulerFactory, Run, ToyFactory}; -pub use stf::{ProvingStf, Stf, ToyInput, ToyOutcome, ToyStf}; +pub use ruler::{Ruler, RulerFactory, Run}; +pub use stf::{ProvingStf, Stf}; pub use structure::{InputBoundary, Position, Quartet, Structure}; diff --git a/cartesi-rollups/node/src/engine/ruler.rs b/cartesi-rollups/node/src/engine/ruler.rs index 01b0d5b98..a7cbf1230 100644 --- a/cartesi-rollups/node/src/engine/ruler.rs +++ b/cartesi-rollups/node/src/engine/ruler.rs @@ -34,7 +34,7 @@ //! start means a broken machine or broken assumptions, and the engine //! panics rather than inventing a transition shape for it. -use super::stf::{ProvingStf, Stf, ToyInput, ToyStf}; +use super::stf::{ProvingStf, Stf}; use super::structure::Structure; use crate::merkle::Digest; use alloy::primitives::U256; @@ -461,21 +461,3 @@ pub trait RulerFactory { type S: Stf; fn ruler_at(&mut self, position: U256) -> Result>; } - -/// Toy factory: each scripted input is one epoch input (payloads are -/// irrelevant to the toy). -pub struct ToyFactory { - pub structure: Structure, - pub script: Vec, -} - -impl RulerFactory for ToyFactory { - type S = ToyStf; - - fn ruler_at(&mut self, position: U256) -> Result> { - let stf = ToyStf::new(self.structure, self.script.clone()); - let mut ruler = Ruler::new(stf, self.structure, self.script.len() as u64); - ruler.advance(position)?; - Ok(ruler) - } -} diff --git a/cartesi-rollups/node/src/engine/spec.rs b/cartesi-rollups/node/src/engine/spec.rs index 79ed57c34..7c0596f3e 100644 --- a/cartesi-rollups/node/src/engine/spec.rs +++ b/cartesi-rollups/node/src/engine/spec.rs @@ -3,19 +3,17 @@ //! The executable leaf-convention specification. //! -//! `oracle_counters` enumerates the whole ruler by brute force, written -//! directly from the documented conventions with none of the engine's -//! machinery. Every test compares engine and cache outputs against it. -//! If the engine and the oracle ever disagree, the conventions are -//! ambiguous or one of them is wrong; either way the spec is doing its -//! job. +//! `oracle_digests` enumerates tiny epochs with literal window, cycle, +//! and slot loops, independently of the ruler's scheduling. The geometry +//! tests compare against that sequence; later cache and proof tests also +//! use reference trees built from the ruler's already-checked runs. use super::cache::{PRECOMPUTE_LEVELS, get_or_compute}; use super::config::EngineConfig; use super::dispute::{DisputeSource, LevelCoords, fold_runs}; -use super::ruler::{RulerFactory, Run, ToyFactory}; -use super::stf::{IDLE_CHURN_TICKS, ToyInput, ToyOutcome, ToyStf}; +use super::ruler::{RulerFactory, Run}; use super::structure::{Quartet, Structure}; +use super::toy::{IDLE_CHURN_TICKS, ToyFactory, ToyInput, ToyOutcome, ToyStf}; use crate::merkle::{Digest, MerkleBuilder, MerkleTree}; use crate::storage::Storage; use alloy::primitives::U256; @@ -252,22 +250,34 @@ fn fully_active_state_is_position_plus_one() { } #[test] -fn mid_span_positioning_matches_oracle() { - // A ruler positioned mid-epoch by replay must continue exactly - // where the oracle says it should. +fn positioning_at_each_slot_matches_oracle() { + // Cover partial uarch spans as well as window and big-cycle boundaries. let structure = S_SMALL; for (name, script) in scripts_for(&structure) { let oracle = oracle_digests(&structure, &script); - let quarter = structure.ruler_span() >> 2; let mut factory = ToyFactory { structure, script: script.clone(), }; - let mut ruler = factory.ruler_at(quarter).unwrap(); - let runs = ruler.collect(quarter * U256::from(3), 0).unwrap(); - let lo = u64::try_from(quarter).unwrap() as usize; - let hi = lo * 3; - assert_eq!(expand(&runs), oracle[lo..hi], "script {name}"); + for start in 0..oracle.len() { + let mut ruler = factory.ruler_at(U256::from(start)).unwrap(); + let expected = if start == 0 { + ToyStf::hash_of(0) + } else { + oracle[start - 1] + }; + assert_eq!( + ruler.state_hash().unwrap(), + expected, + "script {name}, position {start}" + ); + let runs = ruler.collect(structure.ruler_span(), 0).unwrap(); + assert_eq!( + expand(&runs), + oracle[start..], + "script {name}, from {start}" + ); + } } } diff --git a/cartesi-rollups/node/src/engine/stf.rs b/cartesi-rollups/node/src/engine/stf.rs index ad3147559..1e9ce33bd 100644 --- a/cartesi-rollups/node/src/engine/stf.rs +++ b/cartesi-rollups/node/src/engine/stf.rs @@ -11,7 +11,6 @@ //! violations (a feed on a running machine, a ureset off-boundary) //! remain panics - those are engine bugs, not machine conditions. -use super::structure::Structure; use crate::merkle::Digest; use anyhow::Result; @@ -20,9 +19,6 @@ pub trait Stf { /// made of. fn state_hash(&mut self) -> Result; - /// Machine halted. - fn halted(&mut self) -> Result; - /// Machine yielded with RX_ACCEPTED and is awaiting input. fn yielded(&mut self) -> Result; @@ -66,18 +62,7 @@ pub trait Stf { /// big machine directly. Idle cycles do not count: the big machine /// does not advance while yielded or halted (idle uarch spans are /// state-preserving, so skipping them is exact at big boundaries). - /// The default composes the uarch verbs. - fn run_big(&mut self, big_cycles: u64) -> Result { - let mut executed = 0; - while executed < big_cycles && !self.terminal()? && !self.yielded()? { - while !self.uarch_halted()? { - self.ustep()?; - } - self.ureset()?; - executed += 1; - } - Ok(executed) - } + fn run_big(&mut self, big_cycles: u64) -> Result; } /// The proving verbs: each mirrors a plain verb, performing the same @@ -107,240 +92,3 @@ pub trait ProvingStf: Stf { /// access log. fn log_ureset(&mut self) -> Result>; } - -/// How a toy input's processing ends. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum ToyOutcome { - Accept, - Reject, - Halt, -} - -/// Per-input script: how many active usteps each big cycle runs before -/// its uarch halts (the remaining slots repeat, the last slot is the -/// ureset), and how processing ends after the final big cycle. The -/// final big cycle models the yield (or halt) instruction itself, so -/// the machine is yielded (halted) as of that cycle's ureset. -#[derive(Debug, Clone)] -pub struct ToyInput { - pub big_cycles: Vec, - pub outcome: ToyOutcome, -} - -/// Idle churn ticks per idle uarch span: how many usteps the toy's -/// "interpreter" spends noticing the machine is yielded or halted -/// before its uarch halts. The real machine spends a few dozen; one -/// tick keeps toy trees hand-computable while modeling the shape. -pub const IDLE_CHURN_TICKS: u64 = 1; - -/// The toy state-transition function. Its state hash is a counter -/// encoded as bytes32, incremented on every state-changing transition, -/// so on a fully active script the state after transition N is N + 1 -/// (with initial state 0). A revert restores the counter to the -/// window's checkpoint. Idle spans (machine yielded or halted) churn a -/// uarch-local tick that colors the hash without touching the counter; -/// the ureset clears it, restoring the base hash, mirroring the real -/// machine's idle periodicity. This keeps expected trees computable by -/// hand in the spec tests. -#[derive(Debug, Clone)] -pub struct ToyStf { - script: Vec, - - counter: u64, - halted: bool, - yielded: bool, - uarch_halted: bool, - /// Idle churn ticks since the last ureset; nonzero only inside an - /// idle uarch span. - uticks: u64, - - // Current input bookkeeping, valid between feed and yield/halt. - fed: usize, - checkpoint: u64, - outcome: ToyOutcome, - big_cycles: Vec, - current_big_cycle: usize, - usteps_in_big_cycle: u64, -} - -impl ToyStf { - /// A pristine toy at the epoch's start: yielded, awaiting input 0, - /// counter (and thus implicit hash) zero. - pub fn new(structure: Structure, script: Vec) -> Self { - structure.assert_valid(); - for input in &script { - assert!(!input.big_cycles.is_empty(), "input needs a big cycle"); - assert!( - input.big_cycles[0] >= 1, - "big cycle 0 needs an active ustep (the fused feed)" - ); - for &k in &input.big_cycles { - assert!( - k < structure.big_span(), - "usteps must fit before the ureset slot" - ); - } - } - assert!( - IDLE_CHURN_TICKS < structure.big_span() - 1, - "idle churn must fit before the closing slot" - ); - ToyStf { - script, - counter: 0, - halted: false, - yielded: true, - uarch_halted: false, - uticks: 0, - fed: 0, - checkpoint: 0, - outcome: ToyOutcome::Accept, - big_cycles: vec![], - current_big_cycle: 0, - usteps_in_big_cycle: 0, - } - } - - pub fn counter(&self) -> u64 { - self.counter - } - - /// The hash of a base state (no idle churn in flight). - pub fn hash_of(counter: u64) -> Digest { - Self::churned_hash_of(counter, 0) - } - - /// The hash of a state mid-idle-span: the counter colored by the - /// uarch-local churn ticks. - pub fn churned_hash_of(counter: u64, uticks: u64) -> Digest { - let mut data = [0u8; 32]; - data[16..24].copy_from_slice(&uticks.to_be_bytes()); - data[24..].copy_from_slice(&counter.to_be_bytes()); - Digest::from_digest(&data).expect("32 bytes") - } - - fn fixed(&self) -> bool { - self.halted || self.yielded - } -} - -impl Stf for ToyStf { - fn state_hash(&mut self) -> Result { - Ok(Self::churned_hash_of(self.counter, self.uticks)) - } - - fn halted(&mut self) -> Result { - Ok(self.halted) - } - - fn yielded(&mut self) -> Result { - Ok(self.yielded) - } - - fn terminal(&mut self) -> Result { - Ok(self.halted) - } - - fn uarch_halted(&mut self) -> Result { - Ok(self.uarch_halted) - } - - fn feed(&mut self, window: u64) -> Result<()> { - assert!( - self.yielded && !self.halted, - "feed requires a yielded machine" - ); - assert_eq!( - window as usize, self.fed, - "windows feed sequentially from the resume point" - ); - let scripted = self - .script - .get(self.fed) - .expect("toy script must cover every fed input") - .clone(); - self.fed += 1; - self.checkpoint = self.counter; - self.outcome = scripted.outcome; - self.big_cycles = scripted.big_cycles; - self.current_big_cycle = 0; - self.usteps_in_big_cycle = 0; - self.yielded = false; - self.uarch_halted = false; - Ok(()) - } - - fn ustep(&mut self) -> Result<()> { - if self.uarch_halted { - return Ok(()); - } - if self.fixed() { - // Idle churn: uarch-local only. - self.uticks += 1; - if self.uticks == IDLE_CHURN_TICKS { - self.uarch_halted = true; - } - return Ok(()); - } - self.counter += 1; - self.usteps_in_big_cycle += 1; - if self.usteps_in_big_cycle == self.big_cycles[self.current_big_cycle] { - self.uarch_halted = true; - } - Ok(()) - } - - fn ureset(&mut self) -> Result<()> { - if self.fixed() { - // An idle span closes: the churn unwinds, the base state - // returns, and the script does not progress. - assert!(self.uarch_halted, "idle churn must halt the uarch"); - self.uticks = 0; - self.uarch_halted = false; - return Ok(()); - } - assert!( - self.uarch_halted, - "toy script must halt the uarch before the ureset slot" - ); - self.counter += 1; - self.uarch_halted = false; - self.usteps_in_big_cycle = 0; - self.current_big_cycle += 1; - if self.current_big_cycle == self.big_cycles.len() { - // This big cycle was the yield (or halt) instruction. - match self.outcome { - ToyOutcome::Accept => self.yielded = true, - ToyOutcome::Reject => { - self.counter = self.checkpoint; - self.yielded = true; - } - ToyOutcome::Halt => self.halted = true, - } - } - Ok(()) - } -} - -/// Toy witnesses: inert marker bytes over the exact plain-verb state -/// changes, so the Hero's proof path (positioning, agree-state check, -/// shape selection) runs under the toy. Tests may assert which shape -/// was proved from the markers alone. -impl ProvingStf for ToyStf { - fn log_feed(&mut self, window: u64) -> Result> { - if (window as usize) < self.script.len() { - self.feed(window)?; - } - Ok(b"toy-feed;".to_vec()) - } - - fn log_ustep(&mut self) -> Result> { - self.ustep()?; - Ok(b"toy-ustep;".to_vec()) - } - - fn log_ureset(&mut self) -> Result> { - self.ureset()?; - Ok(b"toy-ureset;".to_vec()) - } -} diff --git a/cartesi-rollups/node/src/engine/toy.rs b/cartesi-rollups/node/src/engine/toy.rs new file mode 100644 index 000000000..9ed67b237 --- /dev/null +++ b/cartesi-rollups/node/src/engine/toy.rs @@ -0,0 +1,269 @@ +// (c) Cartesi and individual authors (see AUTHORS) +// SPDX-License-Identifier: Apache-2.0 (see LICENSE) + +//! A scripted machine for geometry and action-preparation unit tests. +//! Inert proof markers allow preparation without real machine witnesses. + +use super::ruler::{Ruler, RulerFactory}; +use super::stf::{ProvingStf, Stf}; +use super::structure::Structure; +use crate::merkle::Digest; +use alloy::primitives::U256; +use anyhow::Result; + +/// How a toy input's processing ends. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ToyOutcome { + Accept, + Reject, + Halt, +} + +/// Per-input script: how many active usteps each big cycle runs before +/// its uarch halts (the remaining slots repeat, the last slot is the +/// ureset), and how processing ends after the final big cycle. The +/// final big cycle models the yield (or halt) instruction itself, so +/// the machine is yielded (halted) as of that cycle's ureset. +#[derive(Debug, Clone)] +pub struct ToyInput { + pub big_cycles: Vec, + pub outcome: ToyOutcome, +} + +/// Idle churn ticks per idle uarch span: how many usteps the toy's +/// "interpreter" spends noticing the machine is yielded or halted +/// before its uarch halts. The real machine spends a few dozen; one +/// tick keeps toy trees hand-computable while modeling the shape. +pub const IDLE_CHURN_TICKS: u64 = 1; + +/// The toy state-transition function. Its state hash is a counter +/// encoded as bytes32, incremented on every state-changing transition, +/// so on a fully active script the state after transition N is N + 1 +/// (with initial state 0). A revert restores the counter to the +/// window's checkpoint. Idle spans (machine yielded or halted) churn a +/// uarch-local tick that colors the hash without touching the counter; +/// the ureset clears it, restoring the base hash, mirroring the real +/// machine's idle periodicity. This keeps expected trees computable by +/// hand in the spec tests. +#[derive(Debug, Clone)] +pub struct ToyStf { + script: Vec, + + counter: u64, + halted: bool, + yielded: bool, + uarch_halted: bool, + /// Idle churn ticks since the last ureset; nonzero only inside an + /// idle uarch span. + uticks: u64, + + // Current input bookkeeping, valid between feed and yield/halt. + fed: usize, + checkpoint: u64, + outcome: ToyOutcome, + big_cycles: Vec, + current_big_cycle: usize, + usteps_in_big_cycle: u64, +} + +impl ToyStf { + /// A pristine toy at the epoch's start: yielded, awaiting input 0, + /// counter (and thus implicit hash) zero. + pub fn new(structure: Structure, script: Vec) -> Self { + structure.assert_valid(); + for input in &script { + assert!(!input.big_cycles.is_empty(), "input needs a big cycle"); + assert!( + input.big_cycles[0] >= 1, + "big cycle 0 needs an active ustep (the fused feed)" + ); + for &k in &input.big_cycles { + assert!( + k < structure.big_span(), + "usteps must fit before the ureset slot" + ); + } + } + assert!( + IDLE_CHURN_TICKS < structure.big_span() - 1, + "idle churn must fit before the closing slot" + ); + ToyStf { + script, + counter: 0, + halted: false, + yielded: true, + uarch_halted: false, + uticks: 0, + fed: 0, + checkpoint: 0, + outcome: ToyOutcome::Accept, + big_cycles: vec![], + current_big_cycle: 0, + usteps_in_big_cycle: 0, + } + } + + /// The hash of a base state (no idle churn in flight). + pub fn hash_of(counter: u64) -> Digest { + Self::churned_hash_of(counter, 0) + } + + /// The hash of a state mid-idle-span: the counter colored by the + /// uarch-local churn ticks. + pub fn churned_hash_of(counter: u64, uticks: u64) -> Digest { + let mut data = [0u8; 32]; + data[16..24].copy_from_slice(&uticks.to_be_bytes()); + data[24..].copy_from_slice(&counter.to_be_bytes()); + Digest::from_digest(&data).expect("32 bytes") + } + + fn fixed(&self) -> bool { + self.halted || self.yielded + } +} + +impl Stf for ToyStf { + fn state_hash(&mut self) -> Result { + Ok(Self::churned_hash_of(self.counter, self.uticks)) + } + + fn yielded(&mut self) -> Result { + Ok(self.yielded) + } + + fn terminal(&mut self) -> Result { + Ok(self.halted) + } + + fn uarch_halted(&mut self) -> Result { + Ok(self.uarch_halted) + } + + fn feed(&mut self, window: u64) -> Result<()> { + assert!( + self.yielded && !self.halted, + "feed requires a yielded machine" + ); + assert_eq!( + window as usize, self.fed, + "windows feed sequentially from the resume point" + ); + let scripted = self + .script + .get(self.fed) + .expect("toy script must cover every fed input") + .clone(); + self.fed += 1; + self.checkpoint = self.counter; + self.outcome = scripted.outcome; + self.big_cycles = scripted.big_cycles; + self.current_big_cycle = 0; + self.usteps_in_big_cycle = 0; + self.yielded = false; + self.uarch_halted = false; + Ok(()) + } + + fn ustep(&mut self) -> Result<()> { + if self.uarch_halted { + return Ok(()); + } + if self.fixed() { + // Idle churn: uarch-local only. + self.uticks += 1; + if self.uticks == IDLE_CHURN_TICKS { + self.uarch_halted = true; + } + return Ok(()); + } + self.counter += 1; + self.usteps_in_big_cycle += 1; + if self.usteps_in_big_cycle == self.big_cycles[self.current_big_cycle] { + self.uarch_halted = true; + } + Ok(()) + } + + fn ureset(&mut self) -> Result<()> { + if self.fixed() { + // An idle span closes: the churn unwinds, the base state + // returns, and the script does not progress. + assert!(self.uarch_halted, "idle churn must halt the uarch"); + self.uticks = 0; + self.uarch_halted = false; + return Ok(()); + } + assert!( + self.uarch_halted, + "toy script must halt the uarch before the ureset slot" + ); + self.counter += 1; + self.uarch_halted = false; + self.usteps_in_big_cycle = 0; + self.current_big_cycle += 1; + if self.current_big_cycle == self.big_cycles.len() { + // This big cycle was the yield (or halt) instruction. + match self.outcome { + ToyOutcome::Accept => self.yielded = true, + ToyOutcome::Reject => { + self.counter = self.checkpoint; + self.yielded = true; + } + ToyOutcome::Halt => self.halted = true, + } + } + Ok(()) + } + + fn run_big(&mut self, big_cycles: u64) -> Result { + let mut executed = 0; + while executed < big_cycles && !self.terminal()? && !self.yielded()? { + while !self.uarch_halted()? { + self.ustep()?; + } + self.ureset()?; + executed += 1; + } + Ok(executed) + } +} + +/// Inert witness bytes over the plain-verb state changes let action +/// preparation tests exercise positioning and agree/post-state checks. +impl ProvingStf for ToyStf { + fn log_feed(&mut self, window: u64) -> Result> { + if (window as usize) < self.script.len() { + self.feed(window)?; + } + Ok(b"toy-feed;".to_vec()) + } + + fn log_ustep(&mut self) -> Result> { + self.ustep()?; + Ok(b"toy-ustep;".to_vec()) + } + + fn log_ureset(&mut self) -> Result> { + self.ureset()?; + Ok(b"toy-ureset;".to_vec()) + } +} + +/// Toy factory: each scripted input is one epoch input (payloads are +/// irrelevant to the toy). +pub struct ToyFactory { + pub structure: Structure, + pub script: Vec, +} + +impl RulerFactory for ToyFactory { + type S = ToyStf; + + fn ruler_at(&mut self, position: U256) -> Result> { + let stf = ToyStf::new(self.structure, self.script.clone()); + let mut ruler = Ruler::new(stf, self.structure, self.script.len() as u64); + ruler.advance(position)?; + Ok(ruler) + } +} diff --git a/cartesi-rollups/node/src/hero/action.rs b/cartesi-rollups/node/src/hero/action.rs index 278f3c033..0ca15410b 100644 --- a/cartesi-rollups/node/src/hero/action.rs +++ b/cartesi-rollups/node/src/hero/action.rs @@ -730,8 +730,9 @@ fn source_error(action: &'static str, tournament: Address, source: anyhow::Error mod tests { use crate::{ engine::{ - LevelCoords, ToyFactory, ToyInput, ToyOutcome, + LevelCoords, spec::{S_SMALL, toy_source}, + toy::{ToyFactory, ToyInput, ToyOutcome}, }, tournament::domain::{ AwaitingChildMatch, BlockDuration, InnerWinner, JoinDisposition, LiveMatch, diff --git a/cartesi-rollups/node/src/hero/actor.rs b/cartesi-rollups/node/src/hero/actor.rs index 45952fb38..f4f8cf467 100644 --- a/cartesi-rollups/node/src/hero/actor.rs +++ b/cartesi-rollups/node/src/hero/actor.rs @@ -7,7 +7,7 @@ use ::log::{debug, error, info}; use crate::{ chain::{Chain, ChainHead}, - engine::{DisputeSource, Positioner, RulerFactory, stf::ProvingStf}, + engine::{DisputeSource, Positioner}, hero::{ action::{PreparedArenaAction, prepare}, context::HeroContext, @@ -59,11 +59,10 @@ impl HeroTick { } } -/// Generic over the ruler factory so action preparation runs against the toy -/// source in unit tests while production uses [`Positioner`]. -pub struct Hero { +/// The production actor owns one epoch's real machine source. +pub struct Hero { arena_sender: Arc, - source: DisputeSource, + source: DisputeSource, epoch: u64, epoch_initial_hash: Digest, root_tournament: Address, @@ -99,12 +98,7 @@ impl Hero { reader, }) } -} -impl Hero -where - F::S: ProvingStf, -{ pub async fn tick(&mut self) -> Result { let (latest_head, foam) = self.reader.fetch_from_root(self.root_tournament).await?; let chain = self.reader.chain().clone(); diff --git a/cartesi-rollups/node/src/hero/context.rs b/cartesi-rollups/node/src/hero/context.rs index 366f730e3..52f8118bd 100644 --- a/cartesi-rollups/node/src/hero/context.rs +++ b/cartesi-rollups/node/src/hero/context.rs @@ -13,7 +13,7 @@ use thiserror::Error; use crate::{ chain::{Chain, ChainHead}, - engine::{DisputeSource, LevelCoords, RulerFactory}, + engine::{DisputeSource, LevelCoords, Positioner}, merkle::Digest, tournament::{ dispute::{ @@ -87,14 +87,14 @@ impl HeroContext { /// Project the Hero's one local path at a caller-supplied chain head. #[allow(clippy::too_many_arguments)] - pub async fn assemble( + pub async fn assemble( chain: &Chain, head: ChainHead, epoch: u64, epoch_initial_hash: Digest, dispute: &Dispute, standings: &HashMap, - source: &mut DisputeSource, + source: &mut DisputeSource, ) -> Result { let root_descriptor = dispute.root().descriptor(); if root_descriptor.initial_hash() != epoch_initial_hash { @@ -287,10 +287,10 @@ fn assemble_snapshots(path: Vec) -> Result( +fn level_material( epoch: u64, descriptor: TournamentDescriptor, - source: &mut DisputeSource, + source: &mut DisputeSource, ) -> Result { let tournament = descriptor.address(); let coords = level_coords(epoch, descriptor)?; diff --git a/docs/computation-hash.md b/docs/computation-hash.md index 3db5364aa..6438500fa 100644 --- a/docs/computation-hash.md +++ b/docs/computation-hash.md @@ -275,3 +275,11 @@ dispute it should have won. The e2e tests cross-check (1) against (2) every epoch (`test/e2e/rollups/test_env.lua`, `epoch_settlement`), and the stf test cases exercise (3) against both. Preserve these cross-checks when refactoring; they are the executable specification of this document. + +The ruler's unit tests use a small scripted machine to enumerate complete +epochs. A literal window/cycle/slot oracle checks stepping and sampling; +cache and proof tests also use trees built from those checked runs. This +separates geometry errors from machine behavior, but does not establish that +the script models Cartesi correctly. The real-machine differentials and +on-chain state-transition tests provide that separate evidence. The scripted +machine and its proof markers are compiled only for unit tests. From 6a6dd9587582ea491bcff80019c17fc1d70ecbf6 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Mon, 7 Sep 2026 09:36:52 -0300 Subject: [PATCH 3/6] refactor(node): contain transition proof preparation in the engine --- cartesi-rollups/node/src/engine/dispute.rs | 38 +++-- .../node/src/engine/machine_stf.rs | 94 ++++++------ cartesi-rollups/node/src/engine/mod.rs | 5 +- cartesi-rollups/node/src/engine/ruler.rs | 4 +- cartesi-rollups/node/src/engine/spec.rs | 47 ++++++ cartesi-rollups/node/src/engine/stf.rs | 19 +-- cartesi-rollups/node/src/engine/toy.rs | 6 +- cartesi-rollups/node/src/hero/action.rs | 136 ++++++------------ cartesi-rollups/node/tests/engine_machine.rs | 83 +++++------ docs/computation-hash.md | 6 + 10 files changed, 219 insertions(+), 219 deletions(-) diff --git a/cartesi-rollups/node/src/engine/dispute.rs b/cartesi-rollups/node/src/engine/dispute.rs index 1da9151f8..875274fb2 100644 --- a/cartesi-rollups/node/src/engine/dispute.rs +++ b/cartesi-rollups/node/src/engine/dispute.rs @@ -1,8 +1,8 @@ // (c) Cartesi and individual authors (see AUTHORS) // SPDX-License-Identifier: Apache-2.0 (see LICENSE) -//! The dispute-facing node source: every merkle node a tournament -//! hero needs, answered by quartet. +//! The dispute-facing computation source: Merkle nodes answered by +//! quartet and transition witnesses checked against both state hashes. //! //! A tournament level's commitment tree lives at a [`LevelCoords`]: //! its root, the node a match contests, bisection children, and @@ -231,18 +231,34 @@ impl DisputeSource { }) } - pub fn factory(&self) -> &F { + #[cfg(test)] + pub(crate) fn factory(&self) -> &F { &self.factory } - /// A ruler positioned at `position`: the machine verb of the - /// facade. This is what proof positioning uses (the disputed - /// leaf's transition witness) and what entering a nested - /// tournament uses to start producing the nested computation - /// hash. Positioning resumes from the boundary store's nearest - /// answer and densifies as it advances. - pub fn machine_at(&mut self, position: U256) -> Result> { - self.factory.ruler_at(position) + /// A transition witness is usable only if replay reaches the agreed + /// state and proving reaches the claimed post-state. Keep both checks + /// with the positioned machine, before returning any witness bytes. + pub fn prove_transition( + &mut self, + position: U256, + expected_pre_state: Digest, + expected_post_state: Digest, + ) -> Result> { + let mut ruler = self.factory.ruler_at(position)?; + let pre_state = ruler.state_hash()?; + ensure!( + pre_state == expected_pre_state, + "epoch {} transition {position}: pre-state {pre_state} differs from expected {expected_pre_state}", + self.epoch + ); + let (proof, post_state) = ruler.prove_transition()?; + ensure!( + post_state == expected_post_state, + "epoch {} transition {position}: post-state {post_state} differs from expected {expected_post_state}", + self.epoch + ); + Ok(proof) } /// Frontier coverage: the quartet sits at or above window diff --git a/cartesi-rollups/node/src/engine/machine_stf.rs b/cartesi-rollups/node/src/engine/machine_stf.rs index 7ce5b7e85..ce2a3f7fe 100644 --- a/cartesi-rollups/node/src/engine/machine_stf.rs +++ b/cartesi-rollups/node/src/engine/machine_stf.rs @@ -12,7 +12,7 @@ use super::dispute::DisputeSource; use super::ruler::{Ruler, RulerFactory}; -use super::stf::{ProvingStf, Stf}; +use super::stf::Stf; use super::structure::Structure; use crate::arithmetic::add_and_clamp; use crate::merkle::Digest; @@ -360,6 +360,51 @@ impl Stf for MachineStf { self.restore_rejected()?; Ok(ran) } + + fn log_feed(&mut self, window: u64) -> Result> { + // The proving path resolves the payload without touching the + // feed cursor or the checkpoint: the machine is spent after + // the proof. + let payload = match &mut self.feeder { + Feeder::Scratch { inputs, .. } => inputs.get(window as usize).cloned(), + Feeder::Store { storage, epoch, .. } => storage + .input(&InputId { + epoch_number: *epoch, + input_index_in_epoch: window, + })? + .map(|input| input.data), + Feeder::Advance { .. } => { + unreachable!("the advance stf collects forward; proving rides the dispute path") + } + }; + match payload { + Some(input) => { + let revert_root = self.machine.root_hash()?; + let cmio_log = self.machine.log_send_cmio_response( + CmioResponseReason::Advance, + &input, + &revert_root, + LogType::default(), + )?; + Ok([Self::encode_da(&input), Self::encode_access_log(&cmio_log)].concat()) + } + None => Ok(Self::encode_da(&[])), + } + } + + fn log_ustep(&mut self) -> Result> { + let log = self.machine.log_step_uarch(LogType::default())?; + self.ucycle += 1; + Ok(Self::encode_access_log(&log)) + } + + fn log_ureset(&mut self) -> Result> { + let log = self.machine.log_reset_uarch(LogType::default())?; + self.ucycle = 0; + let proof = Self::encode_access_log(&log); + self.restore_rejected()?; + Ok(proof) + } } // The chain witness encoding, byte-compatible with what the on-chain @@ -410,53 +455,6 @@ impl MachineStf { } } -impl ProvingStf for MachineStf { - fn log_feed(&mut self, window: u64) -> Result> { - // The proving path resolves the payload without touching the - // feed cursor or the checkpoint: the machine is spent after - // the proof. - let payload = match &mut self.feeder { - Feeder::Scratch { inputs, .. } => inputs.get(window as usize).cloned(), - Feeder::Store { storage, epoch, .. } => storage - .input(&InputId { - epoch_number: *epoch, - input_index_in_epoch: window, - })? - .map(|input| input.data), - Feeder::Advance { .. } => { - unreachable!("the advance stf collects forward; proving rides the dispute path") - } - }; - match payload { - Some(input) => { - let revert_root = self.machine.root_hash()?; - let cmio_log = self.machine.log_send_cmio_response( - CmioResponseReason::Advance, - &input, - &revert_root, - LogType::default(), - )?; - Ok([Self::encode_da(&input), Self::encode_access_log(&cmio_log)].concat()) - } - None => Ok(Self::encode_da(&[])), - } - } - - fn log_ustep(&mut self) -> Result> { - let log = self.machine.log_step_uarch(LogType::default())?; - self.ucycle += 1; - Ok(Self::encode_access_log(&log)) - } - - fn log_ureset(&mut self) -> Result> { - let log = self.machine.log_reset_uarch(LogType::default())?; - self.ucycle = 0; - let proof = Self::encode_access_log(&log); - self.restore_rejected()?; - Ok(proof) - } -} - /// The engine's positioning residue: a work dir, a spawn counter, /// and the store handle they serve. Positions rulers by resuming /// from the boundary store's nearest stored machine and advancing diff --git a/cartesi-rollups/node/src/engine/mod.rs b/cartesi-rollups/node/src/engine/mod.rs index a50e8036e..7a558134d 100644 --- a/cartesi-rollups/node/src/engine/mod.rs +++ b/cartesi-rollups/node/src/engine/mod.rs @@ -23,7 +23,8 @@ //! - [`dispute::DisputeSource`]: the hero-facing face. Tournament //! coordinates map onto quartets ([`dispute::LevelCoords`]), level 0 //! is served from the persisted regime-1 material (window-root rows -//! plus lazy interior folds), and proofs are sibling descents. +//! plus lazy interior folds). It supplies Merkle proofs by sibling +//! descent and transition witnesses checked against pre/post states. //! //! The spec tests compare stepping and sampling against a literal //! leaf sequence. Cache and proof tests also use trees built from those @@ -47,5 +48,5 @@ pub use config::EngineConfig; pub use dispute::{DisputeSource, LevelCoords, fold_runs}; pub use machine_stf::{MachineStf, Positioner}; pub use ruler::{Ruler, RulerFactory, Run}; -pub use stf::{ProvingStf, Stf}; +pub use stf::Stf; pub use structure::{InputBoundary, Position, Quartet, Structure}; diff --git a/cartesi-rollups/node/src/engine/ruler.rs b/cartesi-rollups/node/src/engine/ruler.rs index a7cbf1230..7cde42530 100644 --- a/cartesi-rollups/node/src/engine/ruler.rs +++ b/cartesi-rollups/node/src/engine/ruler.rs @@ -34,7 +34,7 @@ //! start means a broken machine or broken assumptions, and the engine //! panics rather than inventing a transition shape for it. -use super::stf::{ProvingStf, Stf}; +use super::stf::Stf; use super::structure::Structure; use crate::merkle::Digest; use alloy::primitives::U256; @@ -379,7 +379,7 @@ impl Ruler { } } -impl Ruler { +impl Ruler { /// Proves the transition at the current position: the chain /// witness for exactly one of the three shapes the ruler names, /// plus the post-transition state hash. The caller positions the diff --git a/cartesi-rollups/node/src/engine/spec.rs b/cartesi-rollups/node/src/engine/spec.rs index 7c0596f3e..caf6cbfc5 100644 --- a/cartesi-rollups/node/src/engine/spec.rs +++ b/cartesi-rollups/node/src/engine/spec.rs @@ -281,6 +281,53 @@ fn positioning_at_each_slot_matches_oracle() { } } +#[test] +fn checked_transition_proofs_match_oracle_at_each_slot() { + for (name, script) in scripts_for(&S_SMALL) { + let oracle = oracle_digests(&S_SMALL, &script); + let mut source = toy_source(S_SMALL, &script); + let mut pre_state = ToyStf::hash_of(0); + for (position, &post_state) in oracle.iter().enumerate() { + let proof = source + .prove_transition(U256::from(position), pre_state, post_state) + .unwrap_or_else(|error| panic!("script {name}, position {position}: {error}")); + assert!(!proof.is_empty()); + pre_state = post_state; + } + } +} + +#[test] +fn transition_proof_rejects_wrong_expected_states() { + let script = vec![accept(&[2, 1])]; + let oracle = oracle_digests(&S_SMALL, &script); + let mut source = toy_source(S_SMALL, &script); + let position = U256::ONE; + let wrong = Digest::new([0xff; 32]); + + // A bad pre-state takes precedence when both supplied hashes are wrong. + for (pre_state, observed, label) in [ + (wrong, oracle[0], "pre-state"), + (oracle[0], oracle[1], "post-state"), + ] { + let error = source + .prove_transition(position, pre_state, wrong) + .unwrap_err(); + let message = error.to_string(); + assert!(message.contains("epoch 0 transition 1")); + assert!(message.contains(label)); + assert!(message.contains(&wrong.to_string())); + assert!(message.contains(&observed.to_string())); + } + // A failed preparation does not poison the next attempt. + assert!( + !source + .prove_transition(position, oracle[0], oracle[1]) + .unwrap() + .is_empty() + ); +} + #[test] fn cache_root_matches_oracle_tree() -> Result<()> { for structure in [S_DIAGRAM, S_SMALL] { diff --git a/cartesi-rollups/node/src/engine/stf.rs b/cartesi-rollups/node/src/engine/stf.rs index 1e9ce33bd..bd0e96838 100644 --- a/cartesi-rollups/node/src/engine/stf.rs +++ b/cartesi-rollups/node/src/engine/stf.rs @@ -63,26 +63,17 @@ pub trait Stf { /// does not advance while yielded or halted (idle uarch spans are /// state-preserving, so skipping them is exact at big boundaries). fn run_big(&mut self, big_cycles: u64) -> Result; -} -/// The proving verbs: each mirrors a plain verb, performing the same -/// state change while emitting the chain-encoded witness the on-chain -/// state transition consumes. The byte layout is consensus-critical - -/// it must match what prt/contracts' state-transition decodes - and is -/// pinned by a differential test against the prototype proof path plus -/// the stf e2e scenarios, which drive every shape through the chain. -/// -/// Only the real machine's witnesses mean anything to the chain. The -/// toy implements these verbs with inert marker bytes so the proof -/// PATH (positioning, the agree-state check, shape selection) can run -/// under the toy in unit tests; nothing consumes toy bytes. -pub trait ProvingStf: Stf { + // Logged operations apply the same transitions and emit the witness + // encoding consumed by the on-chain state transition. Machine + // differentials and STF e2e tests check that separate contract. + /// The window-opening witness: the data-availability encoding of /// window `window`'s input (empty when the window has none) and, /// when it does, the input-delivery log that also records the /// revert root. The implementation resolves /// the window to its payload, as with [`Stf::feed`]. The fused - /// first ustep is logged separately by [`ProvingStf::log_ustep`]. + /// first ustep is logged separately by [`Stf::log_ustep`]. fn log_feed(&mut self, window: u64) -> Result>; /// One uarch cycle, with its access log. diff --git a/cartesi-rollups/node/src/engine/toy.rs b/cartesi-rollups/node/src/engine/toy.rs index 9ed67b237..d1f941a95 100644 --- a/cartesi-rollups/node/src/engine/toy.rs +++ b/cartesi-rollups/node/src/engine/toy.rs @@ -5,7 +5,7 @@ //! Inert proof markers allow preparation without real machine witnesses. use super::ruler::{Ruler, RulerFactory}; -use super::stf::{ProvingStf, Stf}; +use super::stf::Stf; use super::structure::Structure; use crate::merkle::Digest; use alloy::primitives::U256; @@ -227,11 +227,7 @@ impl Stf for ToyStf { } Ok(executed) } -} -/// Inert witness bytes over the plain-verb state changes let action -/// preparation tests exercise positioning and agree/post-state checks. -impl ProvingStf for ToyStf { fn log_feed(&mut self, window: u64) -> Result> { if (window as usize) < self.script.len() { self.feed(window)?; diff --git a/cartesi-rollups/node/src/hero/action.rs b/cartesi-rollups/node/src/hero/action.rs index 0ca15410b..5b3573b7e 100644 --- a/cartesi-rollups/node/src/hero/action.rs +++ b/cartesi-rollups/node/src/hero/action.rs @@ -10,7 +10,7 @@ use alloy::primitives::{Address, U256}; use thiserror::Error; use crate::{ - engine::{DisputeSource, RulerFactory, stf::ProvingStf}, + engine::{DisputeSource, RulerFactory}, merkle::{Digest, MerkleProof}, tournament::{ MatchID, @@ -152,23 +152,6 @@ pub enum PrepareError { action: &'static str, tournament: Address, }, - #[error( - "positioned machine state {observed} disagrees with sealed agree state {expected} in tournament {tournament}" - )] - AgreeStateMismatch { - tournament: Address, - expected: Digest, - observed: Digest, - }, - #[error( - "proved post-state {observed} disagrees with side {side:?} final state {expected} in tournament {tournament}" - )] - PostStateMismatch { - tournament: Address, - side: MatchSide, - expected: Digest, - observed: Digest, - }, #[error("local source failed while preparing {action} in tournament {tournament}: {source}")] Source { action: &'static str, @@ -181,15 +164,11 @@ pub enum PrepareError { type PrepareResult = std::result::Result; /// Fulfill exactly one intent from one accepted Hero context. -pub fn prepare( +pub fn prepare( intent: HeroIntent, context: &HeroContext, source: &mut DisputeSource, -) -> PrepareResult -where - F: RulerFactory, - F::S: ProvingStf, -{ +) -> PrepareResult { match intent { HeroIntent::Join(intent) => prepare_join(intent, context, source), HeroIntent::ClaimTimeout(intent) => prepare_timeout(intent, context, source), @@ -464,15 +443,11 @@ fn prepare_seal_material( Ok((left_leaf, right_leaf, agree_state_proof)) } -fn prepare_leaf_proof( +fn prepare_leaf_proof( intent: ProofIntent, context: &HeroContext, source: &mut DisputeSource, -) -> PrepareResult -where - F: RulerFactory, - F::S: ProvingStf, -{ +) -> PrepareResult { const ACTION: &str = "prove leaf"; let (snapshot, material) = local_level(context, intent.tournament, intent.commitment)?; let engagement = validate_engagement( @@ -494,32 +469,13 @@ where let (left_node, right_node) = root_children(ACTION, intent.tournament, material, source)?; let divergence = intent.match_state.divergence(); - let mut ruler = source - .machine_at(divergence.coordinate().cycle()) - .map_err(|source| source_error(ACTION, intent.tournament, source))?; - let agree_state = ruler - .state_hash() + let proof = source + .prove_transition( + divergence.coordinate().cycle(), + divergence.agree_state(), + divergence.final_state(intent.side), + ) .map_err(|source| source_error(ACTION, intent.tournament, source))?; - if agree_state != divergence.agree_state() { - return Err(PrepareError::AgreeStateMismatch { - tournament: intent.tournament, - expected: divergence.agree_state(), - observed: agree_state, - }); - } - - let (proof, post_state) = ruler - .prove_transition() - .map_err(|source| source_error(ACTION, intent.tournament, source))?; - let expected_post_state = divergence.final_state(intent.side); - if post_state != expected_post_state { - return Err(PrepareError::PostStateMismatch { - tournament: intent.tournament, - side: intent.side, - expected: expected_post_state, - observed: post_state, - }); - } Ok(PreparedArenaAction::ProveLeaf { tournament: intent.tournament, @@ -732,7 +688,7 @@ mod tests { engine::{ LevelCoords, spec::{S_SMALL, toy_source}, - toy::{ToyFactory, ToyInput, ToyOutcome}, + toy::{ToyFactory, ToyInput, ToyOutcome, ToyStf}, }, tournament::domain::{ AwaitingChildMatch, BlockDuration, InnerWinner, JoinDisposition, LiveMatch, @@ -786,10 +742,6 @@ mod tests { toy_source(S_SMALL, &script()) } - fn initial_hash(source: &mut DisputeSource) -> Digest { - source.machine_at(U256::ZERO).unwrap().state_hash().unwrap() - } - fn descriptor( address: Address, level: u64, @@ -864,7 +816,7 @@ mod tests { ) -> LiveMatch, ) -> Fixture { let mut source = source(); - let initial_hash = initial_hash(&mut source); + let initial_hash = ToyStf::hash_of(0); let descriptor = descriptor(ROOT, 0, kind, initial_hash, U256::ZERO, log2_stride, height); let coords = LevelCoords::new(0, U256::ZERO, log2_stride, height); let commitment = source.node(&coords.root()).unwrap(); @@ -895,7 +847,7 @@ mod tests { fn join_fixture() -> Fixture { let mut source = source(); - let initial_hash = initial_hash(&mut source); + let initial_hash = ToyStf::hash_of(0); let descriptor = descriptor( ROOT, 0, @@ -1041,14 +993,15 @@ mod tests { } fn proof_fixture(side: MatchSide, fault: ProofFault) -> Fixture { - engaged_fixture(LEAF, 0, 7, side, move |source, _coords, descriptor| { + engaged_fixture(LEAF, 0, 7, side, move |_source, _coords, descriptor| { let position = match side { MatchSide::One => U256::ZERO, MatchSide::Two => U256::ONE, }; - let mut ruler = source.machine_at(position).unwrap(); - let actual_agree = ruler.state_hash().unwrap(); - let (_, actual_post) = ruler.prove_transition().unwrap(); + // The script starts with two active usteps, each adding one. + let before = u64::try_from(position).unwrap(); + let actual_agree = ToyStf::hash_of(before); + let actual_post = ToyStf::hash_of(before + 1); let agree_state = if matches!(fault, ProofFault::Agree) { digest(0xc1) } else { @@ -1078,15 +1031,14 @@ mod tests { fn propagation_fixture(side: MatchSide) -> Fixture { let mut source = source(); - let root_initial = initial_hash(&mut source); + let root_initial = ToyStf::hash_of(0); let parent_descriptor = descriptor(ROOT, 0, NON_LEAF, root_initial, U256::ZERO, 3, 4); let parent_coords = LevelCoords::new(0, U256::ZERO, 3, 4); let parent_commitment = source.node(&parent_coords.root()).unwrap(); let opponent = digest(0xb0); let parent_match = id_for(parent_commitment, opponent, side); - let mut ruler = source.machine_at(U256::ZERO).unwrap(); - let agree_state = ruler.state_hash().unwrap(); + let agree_state = root_initial; let parent_final = digest(0xb1); let opponent_final = digest(0xb2); let (final_state_one, final_state_two) = match side { @@ -1341,7 +1293,7 @@ mod tests { } #[test] - fn leaf_proof_checks_agree_and_post_state_in_both_orientations() { + fn leaf_proof_selects_local_state_and_propagates_source_errors() { for side in [MatchSide::One, MatchSide::Two] { let mut fixture = proof_fixture(side, ProofFault::None); let intent = planned(&fixture.context); @@ -1360,30 +1312,28 @@ mod tests { assert_eq!(match_id, fixture.match_id); assert_eq!(left_node.join(&right_node), fixture.commitment); assert!(!proof.is_empty()); + for (fault, state) in [ + (ProofFault::Agree, "pre-state"), + (ProofFault::Post, "post-state"), + ] { + let mut invalid = proof_fixture(side, fault); + let Err(PrepareError::Source { + action, + tournament, + source, + }) = prepare( + planned(&invalid.context), + &invalid.context, + &mut invalid.source, + ) + else { + panic!("invalid {state} must fail proof preparation for {side:?}"); + }; + assert_eq!(action, "prove leaf"); + assert_eq!(tournament, ROOT); + assert!(source.to_string().contains(state)); + } } - - let mut wrong_agree = proof_fixture(MatchSide::One, ProofFault::Agree); - assert!(matches!( - prepare( - planned(&wrong_agree.context), - &wrong_agree.context, - &mut wrong_agree.source - ), - Err(PrepareError::AgreeStateMismatch { .. }) - )); - - let mut wrong_post = proof_fixture(MatchSide::Two, ProofFault::Post); - assert!(matches!( - prepare( - planned(&wrong_post.context), - &wrong_post.context, - &mut wrong_post.source - ), - Err(PrepareError::PostStateMismatch { - side: MatchSide::Two, - .. - }) - )); } #[test] diff --git a/cartesi-rollups/node/tests/engine_machine.rs b/cartesi-rollups/node/tests/engine_machine.rs index 709744526..717261277 100644 --- a/cartesi-rollups/node/tests/engine_machine.rs +++ b/cartesi-rollups/node/tests/engine_machine.rs @@ -16,7 +16,7 @@ mod common; use common::prototype::{MachineCommitment, MachineCommitmentBuilder}; use cartesi_rollups_prt_node::engine::{ - DisputeSource, LevelCoords, MachineStf, Positioner, Quartet, Stf, Structure, + DisputeSource, LevelCoords, MachineStf, Positioner, Quartet, Ruler, Stf, Structure, }; use cartesi_rollups_prt_node::storage::{Input as StorageInput, InputId, Storage}; use common::epoch_data::EpochData; @@ -691,9 +691,8 @@ fn golden_fixtures_hold() { ); } -/// The workstream-4 differential: the ruler-guided proof path -/// (DisputeSource::machine_at + Ruler::prove_transition) must produce -/// byte-identical chain witnesses to the prototype's get_logs, across +/// The checked proof facade must produce byte-identical chain +/// witnesses to the prototype's get_logs, across /// the transition shapes reachable on the echo epoch: the fed window /// start, a plain ustep, a closing slot, and an inputless window /// start (empty data availability). The revert-carrying closing slot @@ -720,23 +719,23 @@ fn prove_transition_matches_prototype_get_logs() { let (_state_dir, storage) = initialized_storage(&image); let work = scratch(); let mut source = DisputeSource::on_store(storage, 0, work.path().to_path_buf()).unwrap(); - let mut ruler = source.machine_at(meta_cycle).unwrap(); - let agree = ruler.state_hash().unwrap(); - let (new_proof, new_next) = ruler.prove_transition().unwrap(); - let dir = scratch(); let db = EpochData::new(inputs.clone(), dir.path().to_path_buf()).unwrap(); + let agree = + MachineInstance::new_rollups_advanced_until(image.to_str().unwrap(), meta_cycle, &db) + .unwrap() + .root_hash() + .unwrap(); let (old_proof, old_next) = MachineInstance::get_logs(image.to_str().unwrap(), 0, agree, meta_cycle, &db).unwrap(); + let new_proof = source + .prove_transition(meta_cycle, agree, old_next) + .unwrap(); assert_eq!( new_proof, old_proof, "proof bytes diverge at {label} (position {meta_cycle})" ); - assert_eq!( - new_next, old_next, - "post-transition hash diverges at {label} (position {meta_cycle})" - ); println!("{label}: {} witness bytes agree", new_proof.len()); } } @@ -745,11 +744,10 @@ fn prove_transition_matches_prototype_get_logs() { /// rejects every input). Three agreements, in dependency order: the /// plain path's closing leaf must be the restored checkpoint (the /// pre-feed state - what the chain restores from the shadow slot); -/// the proving path must report that same post-state (the hero's -/// pre-send check compares it against the commitment, so a prover -/// that reports the discarded rejected state instead can never send -/// winLeafMatch and forfeits by clock); and the witness bytes must -/// match the prototype proof path. +/// the proof facade must validate that same post-state (reporting the +/// discarded rejected state would prevent winLeafMatch and forfeit +/// the dispute by clock); and the witness bytes must match the +/// prototype proof path. #[test] #[ignore = "requires verified echo and yield machine images; run `just test-engine-machine`"] fn revert_closing_slot_restores_the_checkpoint() { @@ -763,25 +761,22 @@ fn revert_closing_slot_restores_the_checkpoint() { let work = scratch(); let mut source = DisputeSource::on_store(storage, 0, work.path().to_path_buf()).unwrap(); - // The agreed pre-state of the window: what the feed checkpoints - // and what the revert must restore. - let pre_feed = { - let mut ruler = source.machine_at(U256::ZERO).unwrap(); - ruler.state_hash().unwrap() - }; - - // Find where the reject lands: feed window 0 and run the big - // machine until the guest yields. The closing slot of that big + // Record what feed checkpoints, then find where the reject lands: + // feed window 0 and run the big machine until the guest yields. + // The closing slot of that big // cycle carries the revert (mirrors stf_revert's oracle-reported // processing_bigs). - let bigs = { - let ruler = source.machine_at(U256::ZERO).unwrap(); - let mut stf = ruler.into_stf(); + let (pre_feed, bigs) = { + let work = scratch(); + let mut stf = MachineStf::load(&image, work.path().to_path_buf()) + .unwrap() + .with_inputs(inputs.clone()); + let pre_feed = stf.state_hash().unwrap(); stf.feed(0).unwrap(); let ran = stf.run_big(u64::MAX).unwrap(); assert!(stf.yielded().unwrap(), "the yield program must yield"); assert!(ran > 0, "the guest must run before yielding"); - ran + (pre_feed, ran) }; let boundary = U256::from(bigs) * big_span; assert!(boundary < (U256::ONE << 44), "input overran a level-0 leaf"); @@ -789,7 +784,12 @@ fn revert_closing_slot_restores_the_checkpoint() { // The built leaf, through the plain path. let built = { - let mut ruler = source.machine_at(closing).unwrap(); + let work = scratch(); + let stf = MachineStf::load(&image, work.path().to_path_buf()) + .unwrap() + .with_inputs(inputs.clone()); + let mut ruler = Ruler::new(stf, structure, inputs.len() as u64); + ruler.advance(closing).unwrap(); let runs = ruler.collect(boundary, 0).unwrap(); runs.last().unwrap().hash }; @@ -798,28 +798,23 @@ fn revert_closing_slot_restores_the_checkpoint() { "the revert must restore the pre-feed state" ); - // The proving path must report the post-state it just proved the - // chain would compute. - let mut ruler = source.machine_at(closing).unwrap(); - let agree = ruler.state_hash().unwrap(); - let (proof, post) = ruler.prove_transition().unwrap(); - assert_eq!( - post, built, - "prove_transition post-state diverges from the built leaf at the revert closing slot" - ); - - // Differential: the prototype proof path agrees on bytes and - // post-state. + // Derive the agreed pre-state independently, then require the + // facade to prove the plain path's restored leaf. let dir = scratch(); let db = EpochData::new(inputs, dir.path().to_path_buf()).unwrap(); + let agree = MachineInstance::new_rollups_advanced_until(image.to_str().unwrap(), closing, &db) + .unwrap() + .root_hash() + .unwrap(); let (old_proof, old_post) = MachineInstance::get_logs(image.to_str().unwrap(), 0, agree, closing, &db).unwrap(); + let proof = source.prove_transition(closing, agree, built).unwrap(); assert_eq!( proof, old_proof, "revert witness bytes diverge from the prototype" ); assert_eq!( - post, old_post, + built, old_post, "post-transition hash diverges from the prototype at the revert closing slot" ); println!( diff --git a/docs/computation-hash.md b/docs/computation-hash.md index 6438500fa..384557679 100644 --- a/docs/computation-hash.md +++ b/docs/computation-hash.md @@ -98,6 +98,12 @@ big-step boundary calls `UArchStep` and then `UArchReset`; every other position calls only `UArchStep`. Each branch requires the access-log proof buffer to be consumed completely before returning its root. +The node's computation source owns transition-proof preparation: it replays +to the disputed position, checks the sealed agree-state hash, generates the +witness, and checks the claimed post-state hash before returning bytes. Hero +selects the position and its side's claimed state from the match. A mismatch +is a local preparation error; no proof action is submitted. + Inside one big cycle, the uarch typically halts long before spending its 2^20 budget. The remaining slots are padded by repeating the halted state hash (`repetitions` in leaf storage), so every big cycle contributes exactly From 7a88e0d6ae9b57c21e3affe0a82e9eca000d2f08 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Mon, 7 Sep 2026 10:06:42 -0300 Subject: [PATCH 4/6] refactor(node): derive advance material from the planned context --- cartesi-rollups/node/src/hero/action.rs | 76 ++++++++++++++---------- cartesi-rollups/node/src/hero/planner.rs | 28 +++------ 2 files changed, 51 insertions(+), 53 deletions(-) diff --git a/cartesi-rollups/node/src/hero/action.rs b/cartesi-rollups/node/src/hero/action.rs index 5b3573b7e..5b276d43e 100644 --- a/cartesi-rollups/node/src/hero/action.rs +++ b/cartesi-rollups/node/src/hero/action.rs @@ -1,10 +1,9 @@ //! Fallible local fulfillment of one pure Hero intent. //! -//! Preparation revalidates the intent against the accepted semantic context, -//! derives every Merkle opening or machine witness, and returns one owned arena -//! action. It performs no provider reads and sends no transaction. In -//! particular, the join bond is resolved only when the prepared join is -//! submitted. +//! Preparation fulfills a decision against the same immutable context that +//! produced it, derives every Merkle opening or machine witness, and returns +//! one owned arena action. It performs no provider reads and sends no +//! transaction. The join bond is resolved when the prepared join is submitted. use alloy::primitives::{Address, U256}; use thiserror::Error; @@ -94,7 +93,7 @@ pub enum PrepareError { #[error("semantic and local-material descriptors disagree for tournament {tournament}")] LevelDescriptorMismatch { tournament: Address }, #[error( - "intent commitment {observed} disagrees with local commitment {expected} in tournament {tournament}" + "commitment {observed} disagrees with local commitment {expected} in tournament {tournament}" )] CommitmentMismatch { tournament: Address, @@ -163,7 +162,7 @@ pub enum PrepareError { type PrepareResult = std::result::Result; -/// Fulfill exactly one intent from one accepted Hero context. +/// Fulfill an intent using the same immutable context that produced it. pub fn prepare( intent: HeroIntent, context: &HeroContext, @@ -253,32 +252,22 @@ fn prepare_advance( source: &mut DisputeSource, ) -> PrepareResult { const ACTION: &str = "advance"; - let (snapshot, material) = local_level(context, intent.tournament, intent.commitment)?; - let engagement = validate_engagement( - ACTION, - intent.tournament, - snapshot, - intent.match_id, - intent.commitment, - intent.side, - )?; - if engagement.live().timeout() != TimeoutDisposition::None - || engagement.live().state() != LiveMatchState::Bisecting(intent.match_state) - || intent.match_state.responder() != intent.side - { + let (snapshot, material) = context_level(context, intent.tournament)?; + let LocalCommitmentStanding::Engaged(engagement) = snapshot.local_standing() else { return Err(PrepareError::IntentStateMismatch { action: ACTION, tournament: intent.tournament, }); - } + }; + let LiveMatchState::Bisecting(state) = engagement.live().state() else { + return Err(PrepareError::IntentStateMismatch { + action: ACTION, + tournament: intent.tournament, + }); + }; - let (left_node, right_node, selected) = unresolved_opening( - ACTION, - intent.tournament, - material, - intent.match_state, - source, - )?; + let (left_node, right_node, selected) = + unresolved_opening(ACTION, intent.tournament, material, state, source)?; let (new_left_node, new_right_node) = source .children(&selected) .map_err(|source| source_error(ACTION, intent.tournament, source))?; @@ -294,7 +283,7 @@ fn prepare_advance( Ok(PreparedArenaAction::Advance { tournament: intent.tournament, - match_id: intent.match_id, + match_id: engagement.match_id(), left_node, right_node, new_left_node, @@ -549,6 +538,22 @@ fn local_level( context: &HeroContext, tournament: Address, commitment: Digest, +) -> PrepareResult<(&SemanticSnapshot, &LevelMaterial)> { + let (snapshot, material) = context_level(context, tournament)?; + let expected = snapshot.local_commitment(); + if commitment != expected { + return Err(PrepareError::CommitmentMismatch { + tournament, + expected, + observed: commitment, + }); + } + Ok((snapshot, material)) +} + +fn context_level( + context: &HeroContext, + tournament: Address, ) -> PrepareResult<(&SemanticSnapshot, &LevelMaterial)> { let snapshot = context .snapshot_at(tournament) @@ -562,11 +567,11 @@ fn local_level( return Err(PrepareError::LevelDescriptorMismatch { tournament }); } let expected = snapshot.local_commitment(); - if material.root() != expected || commitment != expected { + if material.root() != expected { return Err(PrepareError::CommitmentMismatch { tournament, expected, - observed: commitment, + observed: material.root(), }); } Ok((snapshot, material)) @@ -1174,7 +1179,14 @@ mod tests { let HeroIntent::Advance(intent) = planned(&fixture.context) else { panic!("advance fixture must plan an advance"); }; - let state = intent.match_state; + let LocalCommitmentStanding::Engaged(engagement) = + fixture.context.snapshot().local_standing() + else { + panic!("advance fixture must be engaged"); + }; + let LiveMatchState::Bisecting(state) = engagement.live().state() else { + panic!("advance fixture must be bisecting"); + }; let waiting = state.waiting_children(); let PreparedArenaAction::Advance { tournament, diff --git a/cartesi-rollups/node/src/hero/planner.rs b/cartesi-rollups/node/src/hero/planner.rs index afa50a1aa..87b32dc0e 100644 --- a/cartesi-rollups/node/src/hero/planner.rs +++ b/cartesi-rollups/node/src/hero/planner.rs @@ -12,9 +12,9 @@ use crate::{ tournament::{ MatchID, domain::{ - BisectingMatch, BlockDuration, Engagement, InnerEliminationReason, LiveMatchState, - MatchSide, ParentLink, ReadyToSealMatch, SealedLeafMatch, SemanticSnapshot, - TimeoutDisposition, TournamentStanding, + BlockDuration, Engagement, InnerEliminationReason, LiveMatchState, MatchSide, + ParentLink, ReadyToSealMatch, SealedLeafMatch, SemanticSnapshot, TimeoutDisposition, + TournamentStanding, }, }, }; @@ -77,10 +77,6 @@ pub struct TimeoutIntent { #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct AdvanceIntent { pub tournament: Address, - pub match_id: MatchID, - pub commitment: Digest, - pub side: MatchSide, - pub match_state: BisectingMatch, } #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -223,13 +219,7 @@ fn plan_engagement(snapshot: &SemanticSnapshot, engagement: Engagement) -> HeroD match engagement.live().state() { LiveMatchState::Bisecting(match_state) => { plan_responder(match_state.responder(), side, || { - HeroIntent::Advance(AdvanceIntent { - tournament, - match_id, - commitment, - side, - match_state, - }) + HeroIntent::Advance(AdvanceIntent { tournament }) }) } LiveMatchState::ReadyToSealLeaf(match_state) => { @@ -309,9 +299,9 @@ mod tests { use super::*; use crate::tournament::domain::{ - AwaitingChildMatch, EliminationReason, EliminationRecord, InnerWinner, JoinDisposition, - LocalCommitmentStanding, MatchCoordinate, ReadyToSealMatch, RootWinner, SealedDivergence, - TournamentDescriptor, TournamentKind, WaitingChildren, + AwaitingChildMatch, BisectingMatch, EliminationReason, EliminationRecord, InnerWinner, + JoinDisposition, LocalCommitmentStanding, MatchCoordinate, ReadyToSealMatch, RootWinner, + SealedDivergence, TournamentDescriptor, TournamentKind, WaitingChildren, }; fn digest(byte: u8) -> Digest { @@ -650,10 +640,6 @@ mod tests { )), HeroDecision::Act(HeroIntent::Advance(AdvanceIntent { tournament: address(10), - match_id: match_id(), - commitment: digest(1), - side: MatchSide::One, - match_state: bisecting, })) ); From 9f6a1bf9098c616ddaa5fc9c37b7e9cd0d6b6e0d Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Mon, 7 Sep 2026 10:06:52 -0300 Subject: [PATCH 5/6] docs(node): accept cold-start replay within available memory --- docs/node-architecture.md | 32 +++++++++++++++++--------------- 1 file changed, 17 insertions(+), 15 deletions(-) diff --git a/docs/node-architecture.md b/docs/node-architecture.md index cdebbc522..265c2a43c 100644 --- a/docs/node-architecture.md +++ b/docs/node-architecture.md @@ -278,29 +278,24 @@ State and storage: synced, and renamed without replacement, but correctness still relies on exclusive state-directory ownership and no external mutation of committed snapshots. -3. Finalized input and epoch ingestion accumulates the entire unprocessed - block range into in-memory vectors before one database transaction. Range - partitioning limits what each RPC request asks for, but not total backlog - memory or crash replay. A long cold-start backlog should eventually be - committed in bounded block or log chunks. Error handling and observability: -4. Panics and asserts remain on hot paths. The settle-mismatch +3. Panics and asserts remain on hot paths. The settle-mismatch assertions in `src/epoch_manager/mod.rs` deliberately stop on a consensus-critical local/on-chain disagreement. The semantic Hero path now returns observer, context, and fulfillment errors for ordinary invalid observations, but invariant `expect`s remain and still need a dedicated panic-surface audit. -5. Logging is unstructured and inconsistent between crates. -6. Every tournament and settlement request carries the configurable +4. Logging is unstructured and inconsistent between crates. +5. Every tournament and settlement request carries the configurable `15_000_000` gas default. A pool may require balance for `gas_limit * max_fee_per_gas + value`, not expected gas use; join value is therefore additional to the fee envelope. A batch needs enough balance for its cumulative fee envelopes and values, including nested join bonds. Per-verb limits and a calibrated operating funding floor remain pre-mainnet work. -7. The lane does not observe receipts or mined revert reasons. Revert protection +6. The lane does not observe receipts or mined revert reasons. Revert protection at the submission endpoint may reject stale or racing transactions before inclusion, but the node neither requires that service nor detects a deterministic self-authored revert. Because reverted state remains @@ -313,21 +308,28 @@ Error handling and observability: Structure: -8. The reader uses async recursion for dynamic tournament discovery, and the +7. The reader uses async recursion for dynamic tournament discovery, and the Hero's dispute loop runs inside the epoch manager task. Local machine and proof preparation can therefore pin a runtime worker. Moving local dispute work to the blocking lane remains open. -9. Commented-out code blocks kept as reference (the test-scaffolding +8. Commented-out code blocks kept as reference (the test-scaffolding `instance.rs` snapshot logic) and disabled/empty tests. -10. No graceful-shutdown story for in-flight work: a mid-epoch machine run - or mid-dispute reaction is only interrupted at the next poll. +9. No graceful-shutdown story for in-flight work: a mid-epoch machine run + or mid-dispute reaction is only interrupted at the next poll. Design assumptions: -11. Finalized-only persistence. The tournament reader additionally acts on a +10. Finalized-only persistence. The tournament reader additionally acts on a disposable number-range tail and point views at one sampled hash. It does not prove the tail belongs to that hash's ancestry; stale work is safe because mutators revalidate it, and the next tick rebuilds the tail. -12. One node instance per state dir; SQLite WAL is the only cross-thread +11. One node instance per state dir; SQLite WAL is the only cross-thread coordination. Shared state-directory operation is unsupported and has no process lock or recovery protocol. +12. Ingestion holds the application's unprocessed input payloads and epoch + events in memory, including temporary conversion copies, before committing + them with the ingestion watermark. RPC range partitioning does not bound + that total. Operation assumes this backlog fits available RAM and accepts + cold-start and retry costs. A same-state restart resumes from the last + commit; a fresh state directory ingests the application's history again. + Bounded ingestion is warranted only if measured history sizes require it. From 590f3766e632dc0877f7619f266b9303be2d9376 Mon Sep 17 00:00:00 2001 From: gcdepaula Date: Wed, 9 Sep 2026 10:44:21 -0300 Subject: [PATCH 6/6] docs(node): clarify refund completion and diagnostics --- cartesi-rollups/node/src/epoch_manager/recovery.rs | 5 ++++- docs/node-architecture.md | 6 ++++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/cartesi-rollups/node/src/epoch_manager/recovery.rs b/cartesi-rollups/node/src/epoch_manager/recovery.rs index dd9618372..2a79e706b 100644 --- a/cartesi-rollups/node/src/epoch_manager/recovery.rs +++ b/cartesi-rollups/node/src/epoch_manager/recovery.rs @@ -81,7 +81,10 @@ pub async fn plan_recovery( RECOVERABLE | RECOVERED | NO_WINNER => {} TOURNAMENT_RUNNING => tick.complete = false, other => { - log::warn!("undefined bond disposition {other}; keeping candidate inert"); + log::warn!( + "unknown bond disposition {other} for tournament {tournament}; \ + keeping epoch incomplete" + ); tick.complete = false; } } diff --git a/docs/node-architecture.md b/docs/node-architecture.md index 265c2a43c..e9e9d99b5 100644 --- a/docs/node-architecture.md +++ b/docs/node-architecture.md @@ -240,6 +240,12 @@ complete; running and unknown dispositions are not. Restart resumes that cursor. The manager handles settled historical epochs using recovery reads and calls alone, without constructing a Hero or waiting for local execution. +Root settlement implies every linked descendant has finished. Creating a child +pauses the parent match's clocks, and resolving that match requires the child +to finish; the same constraint applies recursively regardless of bond +ownership. The `TOURNAMENT_RUNNING` check therefore adds no separate wait after +finalized settlement. + For the current epoch, each tick combines the Hero's action or one cleanup, an applicable settlement step, and every available bond recovery in one batch. Recovery runs even while the root is contested and retries on the same