diff --git a/.github/workflows/go_build.yml b/.github/workflows/go_build.yml index 50383467f..7b7d01159 100644 --- a/.github/workflows/go_build.yml +++ b/.github/workflows/go_build.yml @@ -17,3 +17,33 @@ jobs: - name: Build run: go build -v ./cmd/mithril + + regression-tests: + runs-on: ubuntu-latest + timeout-minutes: 20 + permissions: + contents: read + env: + GOMAXPROCS: "2" + steps: + - uses: actions/checkout@v3 + + - name: Setup Go + uses: actions/setup-go@v4 + with: + go-version: 1.26.4 + + - name: Voting, checkpoint, streaming and scheduler race regressions + # Run the complete affected package suites, including subprocess crash + # recovery and cancellation tests. + run: >- + go test -race -p 2 -count=1 + ./pkg/alpenglow ./pkg/consensus ./pkg/replay ./pkg/rewards + ./pkg/turbine ./pkg/sigverify ./pkg/blockprod/... + ./cmd/mithril/node ./cmd/mithril/configcmd + + - name: Interpreter differential and memory race regressions + run: go test -race -count=1 ./pkg/sbpf/... + + - name: Sealevel interpreter, syscall and vote ownership regressions + run: go test -race -count=1 ./pkg/sealevel diff --git a/cmd/mithril/configcmd/configcmd.go b/cmd/mithril/configcmd/configcmd.go index 903ff68ca..54fa1952c 100644 --- a/cmd/mithril/configcmd/configcmd.go +++ b/cmd/mithril/configcmd/configcmd.go @@ -158,6 +158,9 @@ authorized_voter_keypair = "" # BLS derivation signer (empty defaults to id authorized_withdrawer_keypair = "" # Authorized withdrawer keypair path (diagnostics only) tpu_quic_bind_addr = "0.0.0.0:8004" advertised_ip = "" # Required only in validator mode; public IP advertised for TPU QUIC +wait_to_vote_slot = 0 # Minimum slot for new votes; does not bypass recovery checks +tpu_max_buffered_transactions = 0 # 0 = 131,072 buffered transactions +block_completion_reserve_ms = 0 # 0 = 75ms local completion/broadcast reserve tpu_sigverify_workers = 0 # 0 = GOMAXPROCS [consensus] @@ -175,6 +178,9 @@ authorized_voter_keypair = "" # Empty defaults to authorized_withdrawer_keypair = "" tpu_quic_bind_addr = "0.0.0.0:8004" advertised_ip = "" # REQUIRED: public IP advertised for TPU QUIC +wait_to_vote_slot = 0 # Minimum slot for new votes; does not bypass recovery checks +tpu_max_buffered_transactions = 0 # 0 = 131,072 buffered transactions +block_completion_reserve_ms = 0 # 0 = 75ms local completion/broadcast reserve tpu_sigverify_workers = 0 [consensus] @@ -261,7 +267,12 @@ max_rps = 8 # Verifier's own RPC budget (never shares the block-fe # ── Replay tuning ──────────────────────────────────────────────────────── [tuning] txpar = 24 # Validator auto-defaults to 2x CPU cores only when unset; explicit 0 = sequential -sigverify_backend = "auto" # auto|r51|generic|stdlib; stdlib uses Go's crypto/ed25519 impl after strict checks. + +[sigverify] +backend = "auto" # auto|r51|generic|stdlib +workers = 0 # 0 = min(2, GOMAXPROCS); explicit value overrides the shared transaction pool +batch_target = 8 # 4 or 8 signature lanes; available work runs immediately +disable_shred_overlap = false # Diagnostic fallback: verify after complete block assembly # ── Mithril's RPC server ───────────────────────────────────────────────── [rpc] diff --git a/cmd/mithril/configcmd/configcmd_test.go b/cmd/mithril/configcmd/configcmd_test.go new file mode 100644 index 000000000..d6905a819 --- /dev/null +++ b/cmd/mithril/configcmd/configcmd_test.go @@ -0,0 +1,28 @@ +package configcmd + +import ( + "strings" + "testing" + + "github.com/spf13/viper" + "github.com/stretchr/testify/require" +) + +func TestStarterConfigSignatureVerification(t *testing.T) { + for _, validator := range []bool{false, true} { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(generateStarterConfig(validator)))) + for _, key := range []string{"validator.wait_to_vote_slot", "validator.tpu_max_buffered_transactions", "validator.block_completion_reserve_ms"} { + require.True(t, v.IsSet(key), key) + require.Zero(t, v.GetInt(key), key) + } + require.Equal(t, "auto", v.GetString("sigverify.backend")) + require.True(t, v.IsSet("sigverify.workers")) + require.Zero(t, v.GetInt("sigverify.workers")) + require.Equal(t, 8, v.GetInt("sigverify.batch_target")) + require.True(t, v.IsSet("sigverify.disable_shred_overlap")) + require.False(t, v.GetBool("sigverify.disable_shred_overlap")) + require.False(t, v.IsSet("tuning.sigverify_backend")) + } +} diff --git a/cmd/mithril/node/config_bool.go b/cmd/mithril/node/config_bool.go new file mode 100644 index 000000000..e779efb61 --- /dev/null +++ b/cmd/mithril/node/config_bool.go @@ -0,0 +1,15 @@ +package node + +import "github.com/spf13/pflag" + +// resolveBoolOption preserves explicit false at either precedence level. +// Defaults use DefValue, not a flag value potentially left by a previous run. +func resolveBoolOption(flag *pflag.Flag, configured bool, configuredValue bool) bool { + if flag != nil && flag.Changed { + return flag.Value.String() == "true" + } + if configured { + return configuredValue + } + return flag != nil && flag.DefValue == "true" +} diff --git a/cmd/mithril/node/config_bool_test.go b/cmd/mithril/node/config_bool_test.go new file mode 100644 index 000000000..ab23313d0 --- /dev/null +++ b/cmd/mithril/node/config_bool_test.go @@ -0,0 +1,38 @@ +package node + +import ( + "github.com/spf13/pflag" + "github.com/spf13/viper" + "github.com/stretchr/testify/require" + "strings" + "testing" +) + +func TestResolveBoolOptionPrecedence(t *testing.T) { + for _, tc := range []struct { + name, toml, cli string + defaultValue, want bool + }{ + {"omitted true default", "", "", true, true}, + {"omitted false default", "", "", false, false}, + {"TOML false", "enabled=false", "", true, false}, + {"TOML true", "enabled=true", "", false, true}, + {"CLI false beats TOML true", "enabled=true", "false", true, false}, + {"CLI true beats TOML false", "enabled=false", "true", false, true}, + {"CLI false without TOML", "", "false", true, false}, + } { + t.Run(tc.name, func(t *testing.T) { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(tc.toml))) + flags := pflag.NewFlagSet("test", pflag.ContinueOnError) + flags.Bool("enabled", tc.defaultValue, "") + if tc.cli != "" { + require.NoError(t, flags.Set("enabled", tc.cli)) + } + require.Equal(t, tc.want, resolveBoolOption(flags.Lookup("enabled"), v.IsSet("enabled"), v.GetBool("enabled"))) + }) + } + require.False(t, resolveBoolOption(nil, false, false)) + require.True(t, resolveBoolOption(nil, true, true)) +} diff --git a/cmd/mithril/node/node.go b/cmd/mithril/node/node.go index c452d494e..c9c964064 100644 --- a/cmd/mithril/node/node.go +++ b/cmd/mithril/node/node.go @@ -74,35 +74,40 @@ var ( }, } - bootstrapMode string // "auto", "snapshot", "new-snapshot", "new-incremental", or "accountsdb" - snapshotArchivePath string - incrementalSnapshotFilename string - accountsPath string - scratchDirectory string - rpcEndpoints []string - cluster string // "alpenglow", "mainnet-beta", "testnet", or "devnet" - legacyGenesisHash string // explicit lineage for pre-binding AccountsDB/ledger artifacts - blockSource string // "turbine", "rpc", or "lightbringer" - lightbringerEndpoint string - repairCatchupMaxGapSlots int // Resume gaps up to this fill via turbine repair instead of RPC (0 = off) - repairMaxRequestsPerSecond int // Repair request-rate ceiling override (0 = adaptive default) - blockRPCFallback bool // Allow RPC block fetch when > repairCatchupMaxGapSlots behind (default false: shreds only) - blockMaxRPS int // Rate limit for block fetching - blockMaxInflight int // Max concurrent block fetch workers - blockTipPollIntervalMs int // Tip poll interval in milliseconds - blockTipSafetyMargin int // Don't fetch within N slots of tip - consensusModeFlag string // raw --consensus-mode value (cobra binding) - consensusMode string // resolved: "verifying" (default) or "validator" - alpenglowObserverBindAddr string - alpenglowMaxMessageBytes int64 - alpenglowBLSDST string - validatorIdentityKeypair string - validatorVoteAccountKeypair string - validatorAuthorizedVoterKeypair string - validatorWithdrawerKeypair string - validatorTPUQUICBind string - validatorAdvertisedIP string - validatorSigverifyWorkers int + bootstrapMode string // "auto", "snapshot", "new-snapshot", "new-incremental", or "accountsdb" + snapshotArchivePath string + incrementalSnapshotFilename string + accountsPath string + scratchDirectory string + rpcEndpoints []string + cluster string // "alpenglow", "mainnet-beta", "testnet", or "devnet" + legacyGenesisHash string // explicit lineage for pre-binding AccountsDB/ledger artifacts + blockSource string // "turbine", "rpc", or "lightbringer" + lightbringerEndpoint string + repairCatchupMaxGapSlots int // Resume gaps up to this fill via turbine repair instead of RPC (0 = off) + repairMaxRequestsPerSecond int // Repair request-rate ceiling override (0 = adaptive default) + blockRPCFallback bool // Allow RPC block fetch when > repairCatchupMaxGapSlots behind (default false: shreds only) + blockMaxRPS int // Rate limit for block fetching + blockMaxInflight int // Max concurrent block fetch workers + blockTipPollIntervalMs int // Tip poll interval in milliseconds + blockTipSafetyMargin int // Don't fetch within N slots of tip + consensusModeFlag string // raw --consensus-mode value (cobra binding) + consensusMode string // resolved: "verifying" (default) or "validator" + alpenglowObserverBindAddr string + alpenglowMaxMessageBytes int64 + alpenglowBLSDST string + validatorIdentityKeypair string + validatorVoteAccountKeypair string + validatorAuthorizedVoterKeypair string + validatorWithdrawerKeypair string + validatorTPUQUICBind string + validatorAdvertisedIP string + validatorSigverifyWorkers int + validatorWaitToVoteSlot uint64 + validatorReservedHistory bool + validatorInitializeReservation bool + validatorCompletionReserveMs int + validatorMaxBufferedTransactions int // Mode thresholds blockNearTipThreshold int // Enter near-tip when gap <= this @@ -121,6 +126,9 @@ var ( pprofPort int64 blockstorePath string txParallelism int64 + // streamingMaxOpenMs is --streaming-max-open-ms; resolved into + // replay.StreamingExecutionCfg.MaxOpenAge with the other [replay] keys. + streamingMaxOpenMs int debugTxs []string debugAcctWrites []string @@ -533,6 +541,14 @@ func init() { // [replay] section flags Run.Flags().Int64Var(&txParallelism, "txpar", 0, "Transaction execution workers (>0 enables topsort parallelism; explicit 0 is sequential; unset validator mode defaults to 2x CPU cores)") Run.Flags().Int64Var(&numReplaySlots, "num-slots", 0, "Number of slots to replay (0 = run continuously)") + Run.Flags().BoolVar(&replay.StreamingExecutionCfg.Enabled, "streaming-execution", false, + "Execute Turbine blocks while their shreds arrive (Alpenglow validator/verifying modes only; the complete block remains authoritative and any mismatch falls back to whole-block execution)") + Run.Flags().IntVar(&replay.StreamingExecutionCfg.Workers, "streaming-workers", 0, + "Streaming execution workers per transaction group (0 = min(txpar, 4))") + Run.Flags().IntVar(&replay.StreamingExecutionCfg.MinGroupBatches, "streaming-min-group-batches", 0, + "Contiguous decoded batches to accumulate before a streaming group executes (0 or 1 = execute as batches arrive)") + Run.Flags().IntVar(&streamingMaxOpenMs, "streaming-max-open-ms", 0, + "Discard a streaming bank whose block has not completed after this many milliseconds (0 = 2000)") Run.Flags().Int64VarP(&endSlot, "end-slot", "e", -1, "Block at which to stop replaying, inclusive (-1 = run continuously)") // [consensus] section flags @@ -547,6 +563,11 @@ func init() { Run.Flags().StringVar(&validatorTPUQUICBind, "tpu-quic-bind-addr", "", "Validator TPU QUIC listen address (default 0.0.0.0:8004)") Run.Flags().StringVar(&validatorAdvertisedIP, "validator-advertised-ip", "", "Public IP advertised for validator TPU QUIC") Run.Flags().IntVar(&validatorSigverifyWorkers, "tpu-sigverify-workers", 0, "TPU signature verification workers (0 = GOMAXPROCS)") + Run.Flags().BoolVar(&validatorReservedHistory, "reserved-vote-history", false, "Use durable signing reservations with unsynchronized per-vote history writes") + Run.Flags().BoolVar(&validatorInitializeReservation, "initialize-vote-reservation", false, "Enroll complete synchronous vote history in reserved mode (one-time migration)") + Run.Flags().Uint64Var(&validatorWaitToVoteSlot, "wait-to-vote-slot", 0, "Do not cast new votes below this slot; the automatic startup cutoff still applies (0 = automatic only)") + Run.Flags().IntVar(&validatorCompletionReserveMs, "leader-completion-reserve-ms", 0, "Time reserved for leader finalization and broadcast (0 = 75ms default; tune from measured completion times)") + Run.Flags().IntVar(&validatorMaxBufferedTransactions, "tpu-max-buffered-transactions", 0, "Maximum queued TPU transactions (0 = 131072 default)") // [tuning] section flags Run.Flags().Uint64Var(¶mArenaSizeMB, "param-arena-size-mb", 512, "Size in MB for serialized parameter arena (0 to disable)") @@ -560,6 +581,12 @@ func init() { Run.Flags().StringVar(&snapshot.SnapshotIndexTempDir, "snapshot-index-temp-dir", "", "Optional directory for snapshot index shard logs/SST staging") Run.Flags().StringVar(&sigverify.Cfg.Backend, "sigverify-backend", sigverify.Defaults().Backend, "ed25519 verification backend: auto|r51|generic|stdlib") + Run.Flags().IntVar(&sigverify.Cfg.Workers, "sigverify-workers", 0, + "Turbine transaction signature verification workers (0 = min(2, GOMAXPROCS))") + Run.Flags().IntVar(&sigverify.Cfg.BatchTarget, "sigverify-batch-target", sigverify.Defaults().BatchTarget, + "Turbine transaction signature batch target: 4 or 8 (available short batches run immediately)") + Run.Flags().BoolVar(&sigverify.Cfg.DisableShredOverlap, "sigverify-disable-shred-overlap", false, + "Defer Turbine transaction decoding and signature verification until all block shreds arrive") Run.Flags().BoolVar(&sbpf.UsePool, "use-pool", true, "Disable to allocate fresh slices") Run.Flags().IntVar(&accountsdb.StoreAccountsWorkers, "store-accounts-workers", 128, "Number of workers to write account updates") Run.Flags().IntVar(&accountsdb.ProgramCacheMaxMB, "program-cache-max-mb", accountsdb.DefaultProgramCacheMaxMB, "Maximum approximate SBPF program cache size in MiB") @@ -639,6 +666,11 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { if err := config.InitConfig(); err != nil { return err } + if slot, err := configuredWaitToVoteSlot(cmd); err != nil { + return err + } else { + validatorWaitToVoteSlot = slot + } // Check if a CLI flag was explicitly set by the user flagChanged := func(name string) bool { @@ -721,14 +753,9 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { return 0 } - // Helper to get bool: CLI flag if explicitly set, otherwise TOML config + // Match numeric options: explicit CLI, configured value, then flag default. getBool := func(cliKey, tomlKey string) bool { - if flagChanged(cliKey) { - if f := cmd.Flags().Lookup(cliKey); f != nil { - return f.Value.String() == "true" - } - } - return config.GetBool(tomlKey) + return resolveBoolOption(cmd.Flags().Lookup(cliKey), config.IsSet(tomlKey), config.GetBool(tomlKey)) } // Helper to get string slice: CLI flag if explicitly set, otherwise TOML config @@ -863,6 +890,14 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { } validatorAdvertisedIP = getString("validator-advertised-ip", "validator.advertised_ip") validatorSigverifyWorkers = getInt("tpu-sigverify-workers", "validator.tpu_sigverify_workers") + validatorCompletionReserveMs = getInt("leader-completion-reserve-ms", "validator.block_completion_reserve_ms") + validatorMaxBufferedTransactions = getInt("tpu-max-buffered-transactions", "validator.tpu_max_buffered_transactions") + if validatorMaxBufferedTransactions < 0 { + return fmt.Errorf("TPU maximum buffered transactions must be nonnegative") + } + if validatorCompletionReserveMs < 0 || validatorCompletionReserveMs >= int(blockprod.AlpenglowSlotDuration/time.Millisecond) { + return fmt.Errorf("leader completion reserve must be 0 (default) or between 1 and 199 milliseconds") + } // [block] section blockSource = getString("block-source", "block.source") @@ -1127,13 +1162,27 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { // later: narya pins its backend on first use, and selecting it explicitly // doubles as a startup health check, so a machine that cannot run the // requested backend fails now instead of at the first block. - sigverify.Cfg.Backend = getString("sigverify-backend", "tuning.sigverify_backend") + backendKey := "sigverify.backend" + if !config.IsSet(backendKey) { + backendKey = "tuning.sigverify_backend" // older configuration files + } + sigverify.Cfg.Backend = getString("sigverify-backend", backendKey) + sigverify.Cfg.Workers = getInt("sigverify-workers", "sigverify.workers") + sigverify.Cfg.BatchTarget = getInt("sigverify-batch-target", "sigverify.batch_target") + sigverify.Cfg.DisableShredOverlap = getBool("sigverify-disable-shred-overlap", "sigverify.disable_shred_overlap") resolved, err := sigverify.Configure(sigverify.Cfg) if err != nil { - return fmt.Errorf("tuning.sigverify_backend: %w", err) + return fmt.Errorf("signature verification configuration: %w", err) } resolvedSigverifyBackend = resolved sbpf.UsePool = getBool("use-pool", "tuning.use_pool") + // [tuning] streaming execution (off by default; Alpenglow turbine only). + replay.StreamingExecutionCfg.Enabled = getBool("streaming-execution", "tuning.streaming_execution") + replay.StreamingExecutionCfg.Workers = getInt("streaming-workers", "tuning.streaming_workers") + replay.StreamingExecutionCfg.MinGroupBatches = getInt("streaming-min-group-batches", "tuning.streaming_min_group_batches") + if ms := getInt("streaming-max-open-ms", "tuning.streaming_max_open_ms"); ms > 0 { + replay.StreamingExecutionCfg.MaxOpenAge = time.Duration(ms) * time.Millisecond + } accountsdb.StoreAccountsWorkers = getInt("store-accounts-workers", "tuning.store_accounts_workers") accountsdb.ProgramCacheMaxMB = getInt("program-cache-max-mb", "tuning.program_cache_max_mb") if accountsdb.ProgramCacheMaxMB <= 0 { @@ -2642,52 +2691,9 @@ postBootstrap: } global.SeedWallClockSlot(wallClockSeed) startupWallSlot := global.WallClockSlot() - waitToVoteSlot := startupWallSlot - startupWallSlot%alpenglow.LeaderWindowSlots - if waitToVoteSlot <= math.MaxUint64-2*alpenglow.LeaderWindowSlots { - waitToVoteSlot += 2 * alpenglow.LeaderWindowSlots - } else { - waitToVoteSlot = math.MaxUint64 - } - mlog.Log.Infof("ALPENGLOW voting startup watermark: wall_clock=%d wait_to_vote=%d", startupWallSlot, waitToVoteSlot) + waitToVoteSlot := effectiveWaitToVoteSlot(startupWallSlot, validatorWaitToVoteSlot) + mlog.Log.Infof("ALPENGLOW voting startup watermark: wall_clock=%d configured_wait_to_vote=%d wait_to_vote=%d", startupWallSlot, validatorWaitToVoteSlot, waitToVoteSlot) - identityPubkey := solana.PrivateKey(validatorIdentity).PublicKey() - if err := consensusEngine.EnableVoting(consensusengine.VotingConfig{ - Identity: validatorIdentity, - AuthorizedVoter: validatorAuthorizedVoter, - VoteAccount: validatorVoteAccount, - HistoryDir: blockstorePath, - EpochForSlot: epochSchedule.GetEpoch, - SlotDuration: blockprod.AlpenglowSlotDuration, - WaitToVoteSlot: waitToVoteSlot, - ReadyToVote: func(slot uint64) bool { - wallSlot := global.WallClockSlot() - if liveSlot, ok := consensusEngine.AlpenglowLiveSlot(); ok { - wallSlot = liveSlot - } - return slot >= wallSlot || wallSlot-slot <= alpenglow.LeaderWindowSlots - }, - Peers: func(validators []alpenglow.ValidatorStake) []alpenglow.VotorPeer { - peers := make([]alpenglow.VotorPeer, 0, len(validators)) - seen := make(map[solana.PublicKey]struct{}, len(validators)) - for _, validator := range validators { - if validator.Stake == 0 || validator.NodePubkey == identityPubkey { - continue - } - addr, ok := sharedGossip.LookupAlpenglow(validator.NodePubkey) - if !ok { - continue - } - if _, duplicate := seen[validator.NodePubkey]; duplicate { - continue - } - seen[validator.NodePubkey] = struct{}{} - peers = append(peers, alpenglow.VotorPeer{Identity: validator.NodePubkey, Addr: addr}) - } - return peers - }, - }); err != nil { - klog.Fatalf("enable Alpenglow voting: %v", err) - } broadcaster, err := turbine.NewTurbineBroadcaster(turbine.TurbineBroadcasterConfig{ Self: solana.PrivateKey(validatorIdentity).PublicKey(), Peers: sharedGossip, @@ -2705,7 +2711,9 @@ postBootstrap: defer broadcaster.Close() controller := blockprod.NewController() - topicSink := scheduler.New(controller) + topicSink := scheduler.NewWithConfig(controller, scheduler.Config{ + FeatureSource: replay.ChainTipFeatures, MaxBufferedTransactions: validatorMaxBufferedTransactions, + }) topicSink.Start(ctx) defer topicSink.Stop() tpuCfg := tpu.DefaultConfig() @@ -2744,6 +2752,50 @@ postBootstrap: mlog.Log.Warnf("validator gossip TPU advertisement: %v", err) } + // Bind and validate local transports before consuming the durable clean + // voting marker. Startup configuration failures must not force recovery. + identityPubkey := solana.PrivateKey(validatorIdentity).PublicKey() + if err := consensusEngine.EnableVoting(consensusengine.VotingConfig{ + Identity: validatorIdentity, + AuthorizedVoter: validatorAuthorizedVoter, + VoteAccount: validatorVoteAccount, + HistoryDir: blockstorePath, + ReservedHistory: validatorReservedHistory, + InitializeVoteReservation: validatorInitializeReservation, + Genesis: solana.MustHashFromBase58(networkGenesisHash), + EpochForSlot: epochSchedule.GetEpoch, + SlotDuration: blockprod.AlpenglowSlotDuration, + WaitToVoteSlot: waitToVoteSlot, + ReadyToVote: func(slot uint64) bool { + wallSlot := global.WallClockSlot() + if liveSlot, ok := consensusEngine.AlpenglowLiveSlot(); ok { + wallSlot = liveSlot + } + return slot >= wallSlot || wallSlot-slot <= alpenglow.LeaderWindowSlots + }, + Peers: func(validators []alpenglow.ValidatorStake) []alpenglow.VotorPeer { + peers := make([]alpenglow.VotorPeer, 0, len(validators)) + seen := make(map[solana.PublicKey]struct{}, len(validators)) + for _, validator := range validators { + if validator.Stake == 0 || validator.NodePubkey == identityPubkey { + continue + } + addr, ok := sharedGossip.LookupAlpenglow(validator.NodePubkey) + if !ok { + continue + } + if _, duplicate := seen[validator.NodePubkey]; duplicate { + continue + } + seen[validator.NodePubkey] = struct{}{} + peers = append(peers, alpenglow.VotorPeer{Identity: validator.NodePubkey, Addr: addr}) + } + return peers + }, + }); err != nil { + klog.Fatalf("enable Alpenglow voting: %v", err) + } + rewardBuilder := rewardcerts.NewBuilder(rewardcerts.BuilderConfig{ RootSlot: global.Slot, BeforeBuild: consensusEngine.FlushAlpenglowRewardVotes, @@ -2752,14 +2804,15 @@ postBootstrap: leaderStop := make(chan struct{}) leaderDone := make(chan struct{}) leaderLoop := blockprod.NewLeaderLoop(blockprod.LeaderLoopConfig{ - Controller: controller, - Identity: solana.PrivateKey(validatorIdentity), - AccountsDb: accountsDb, - Broadcaster: broadcaster, - ShredVersion: uint16(turbineShredVersion), - EpochSchedule: epochSchedule, - AlpenglowClock: true, - SlotDuration: blockprod.AlpenglowSlotDuration, + Controller: controller, + Identity: solana.PrivateKey(validatorIdentity), + AccountsDb: accountsDb, + Broadcaster: broadcaster, + ShredVersion: uint16(turbineShredVersion), + EpochSchedule: epochSchedule, + AlpenglowClock: true, + SlotDuration: blockprod.AlpenglowSlotDuration, + CompletionReserve: time.Duration(validatorCompletionReserveMs) * time.Millisecond, ParentContext: func(slot uint64) blockprod.ParentContext { tip := replay.ChainTipParentContext() // Blockprod owns the replay-readiness rule. In particular, the first @@ -2796,6 +2849,7 @@ postBootstrap: } }, ProductionParent: consensusEngine.AlpenglowBlockProductionParent, + CanSignSlot: consensusEngine.AlpenglowCanSignLeaderSlot, CurrentSlot: func() uint64 { if slot, ok := consensusEngine.AlpenglowLiveSlot(); ok { return slot @@ -3214,6 +3268,8 @@ func printStartupInfo(commandName string) { } fmt.Printf(" Sigverify: %s%s%s %s(%s)%s\n", green, resolvedSigverifyBackend, reset, dim, sigverifyDesc, reset) + fmt.Printf(" workers=%d batch_target=%d shred_overlap=%t\n", + sigverify.TransactionWorkers(), sigverify.TransactionBatchTarget(), !sigverify.Cfg.DisableShredOverlap) } // Load state file for detailed info (only show for modes that use existing AccountsDB) diff --git a/cmd/mithril/node/sigverify_reporter.go b/cmd/mithril/node/sigverify_reporter.go index d374d93b9..597a3c116 100644 --- a/cmd/mithril/node/sigverify_reporter.go +++ b/cmd/mithril/node/sigverify_reporter.go @@ -42,7 +42,8 @@ func startSigverifyReporter(ctx context.Context) { // against, and so the resolved backend is recorded even on a node that // exits before the first tick. previous := sigverify.Stats() - mlog.NamedFilef("sigverify", "startup: %s", previous) + mlog.NamedFilef("sigverify", "startup: %s workers=%d batch_target=%d shred_overlap=%t", previous, + sigverify.TransactionWorkers(), sigverify.TransactionBatchTarget(), !sigverify.Cfg.DisableShredOverlap) for { select { diff --git a/cmd/mithril/node/vote_startup.go b/cmd/mithril/node/vote_startup.go new file mode 100644 index 000000000..547e55194 --- /dev/null +++ b/cmd/mithril/node/vote_startup.go @@ -0,0 +1,40 @@ +package node + +import ( + "fmt" + "math" + "strconv" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/Overclock-Validator/mithril/pkg/config" + "github.com/spf13/cobra" +) + +func configuredWaitToVoteSlot(cmd *cobra.Command) (uint64, error) { + if flag := cmd.Flags().Lookup("wait-to-vote-slot"); flag != nil && flag.Changed { + return cmd.Flags().GetUint64("wait-to-vote-slot") + } + const key = "validator.wait_to_vote_slot" + if !config.IsSet(key) { + return 0, nil + } + // Unlike GetUint64, parsing explicitly must not turn an invalid operator + // cutoff into zero and silently remove the requested voting restriction. + slot, err := strconv.ParseUint(config.GetString(key), 10, 64) + if err != nil { + return 0, fmt.Errorf("%s must be an unsigned 64-bit slot: %w", key, err) + } + return slot, nil +} + +// The operator cutoff can postpone voting but cannot weaken the existing +// startup guard. Equality permits voting, subject to all other Votor checks. +func effectiveWaitToVoteSlot(startupWallSlot, configured uint64) uint64 { + automatic := startupWallSlot - startupWallSlot%alpenglow.LeaderWindowSlots + if automatic <= math.MaxUint64-2*alpenglow.LeaderWindowSlots { + automatic += 2 * alpenglow.LeaderWindowSlots + } else { + automatic = math.MaxUint64 + } + return max(automatic, configured) +} diff --git a/cmd/mithril/node/vote_startup_test.go b/cmd/mithril/node/vote_startup_test.go new file mode 100644 index 000000000..d86a02528 --- /dev/null +++ b/cmd/mithril/node/vote_startup_test.go @@ -0,0 +1,73 @@ +package node + +import ( + "math" + "strings" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/config" + "github.com/spf13/cobra" + "github.com/spf13/viper" + "github.com/stretchr/testify/require" +) + +func TestConfiguredWaitToVoteSlot(t *testing.T) { + for _, tc := range []struct { + name, toml, cli string + want uint64 + invalid bool + }{ + {name: "default"}, + {name: "toml", toml: "wait_to_vote_slot = 1234", want: 1234}, + {name: "cli wins", toml: "wait_to_vote_slot = 1234", cli: "5678", want: 5678}, + {name: "explicit zero wins", toml: "wait_to_vote_slot = 1234", cli: "0"}, + {name: "maximum CLI", cli: "18446744073709551615", want: math.MaxUint64}, + {name: "negative TOML", toml: "wait_to_vote_slot = -1", invalid: true}, + {name: "fractional TOML", toml: "wait_to_vote_slot = 1.5", invalid: true}, + {name: "malformed TOML value", toml: `wait_to_vote_slot = "oops"`, invalid: true}, + {name: "empty TOML value", toml: `wait_to_vote_slot = ""`, invalid: true}, + {name: "overflow TOML value", toml: `wait_to_vote_slot = "18446744073709551616"`, invalid: true}, + {name: "negative CLI", cli: "-1", invalid: true}, + {name: "overflow CLI", cli: "18446744073709551616", invalid: true}, + } { + t.Run(tc.name, func(t *testing.T) { + viper.Reset() + t.Cleanup(viper.Reset) + config.ApplyDefaults(viper.GetViper()) + viper.SetConfigType("toml") + require.NoError(t, viper.ReadConfig(strings.NewReader("[validator]\n"+tc.toml))) + cmd := &cobra.Command{} + cmd.Flags().Uint64("wait-to-vote-slot", 0, "") + var err error + if tc.cli != "" { + err = cmd.Flags().Set("wait-to-vote-slot", tc.cli) + } + var got uint64 + if err == nil { + got, err = configuredWaitToVoteSlot(cmd) + } + if tc.invalid { + require.Error(t, err) + return + } + require.NoError(t, err) + require.Equal(t, tc.want, got) + }) + } + require.NotNil(t, Run.Flags().Lookup("wait-to-vote-slot")) +} + +func TestEffectiveWaitToVoteSlot(t *testing.T) { + for _, tc := range []struct{ startup, configured, want uint64 }{ + {100, 0, 108}, + {103, 0, 108}, + {103, 104, 108}, + {103, 108, 108}, + {103, 123, 123}, // Operator cutoff need not align with a leader window. + {103, math.MaxUint64, math.MaxUint64}, + {math.MaxUint64 - 7, 0, math.MaxUint64}, + {math.MaxUint64, 0, math.MaxUint64}, + } { + require.Equal(t, tc.want, effectiveWaitToVoteSlot(tc.startup, tc.configured), "%+v", tc) + } +} diff --git a/cmd/repair-sim/main.go b/cmd/repair-sim/main.go new file mode 100644 index 000000000..5eebf05e9 --- /dev/null +++ b/cmd/repair-sim/main.go @@ -0,0 +1,137 @@ +// repair-sim runs deterministic, single-node Turbine repair scenarios. +package main + +import ( + "encoding/json" + "flag" + "fmt" + "os" + "os/exec" + "runtime" + "strings" + "time" + + "github.com/Overclock-Validator/mithril/pkg/turbine/repairsim" +) + +type environment struct { + GoVersion string `json:"go_version"` + GOOS string `json:"goos"` + GOARCH string `json:"goarch"` + CPU string `json:"cpu"` +} + +type report struct { + Environment environment `json:"environment"` + Ledger repairsim.LedgerConfig `json:"ledger"` + Network repairsim.Config `json:"network"` + LedgerGenerationWall time.Duration `json:"ledger_generation_wall_ns"` + Result repairsim.Result `json:"result"` +} + +func main() { + var ( + scenarioFlag = flag.String("scenario", string(repairsim.ScenarioNearTip), "near-tip or deep-catchup") + slots = flag.Int("slots", 200, "number of deterministic slots") + fecSets = flag.Int("fec-sets", 4, "FEC sets generated per slot") + entries = flag.Int("entries", 0, "entries per slot (0 derives an exact FEC count)") + seed = flag.Int64("seed", 1, "deterministic content and network seed") + availability = flag.String("availability", "", "complete, near-loss, sparse, or mixed") + repair = flag.Bool("repair", true, "enable repair requests") + latency = flag.Duration("repair-latency", 20*time.Millisecond, "synthetic one-way response latency") + jitter = flag.Duration("repair-jitter", 2*time.Millisecond, "deterministic +/- response jitter") + loss = flag.Float64("packet-loss", 0, "repair response loss probability [0,1]") + duplicates = flag.Float64("duplicates", 0.02, "duplicate response probability [0,1]") + bandwidth = flag.Int64("repair-bandwidth", 100*1024*1024, "synthetic repair bytes/sec (0 is unlimited)") + concurrent = flag.Int("max-concurrent", 256, "maximum outstanding repair shreds") + corrupt = flag.Int("corrupt-responses", 0, "corrupt the first N repair responses") + naturalLate = flag.Bool("natural-late", true, "schedule selected late live shreds during repair") + spoolDir = flag.String("spool-dir", "", "persistent shred-spool directory (empty uses a temporary directory)") + cpuLabel = flag.String("cpu-label", "", "explicit CPU label when platform discovery is unavailable") + output = flag.String("output", "", "write JSON to this file instead of stdout") + includeTrace = flag.Bool("trace", true, "include the logical event trace in JSON") + ) + flag.Parse() + + scenario := repairsim.Scenario(*scenarioFlag) + network := repairsim.DefaultConfig(scenario) + network.Availability = repairsim.Availability(*availability) + network.RepairEnabled = *repair + network.RepairLatency = *latency + network.RepairJitter = *jitter + network.PacketLoss = *loss + network.DuplicateProbability = *duplicates + network.BandwidthBytesPerSec = *bandwidth + network.MaxConcurrent = *concurrent + network.CorruptResponses = *corrupt + network.NaturalLateShreds = *naturalLate + network.CollectTrace = *includeTrace + network.Seed = *seed + network.SpoolDir = *spoolDir + if network.Availability == "" { + network.Availability = repairsim.DefaultConfig(scenario).Availability + } + + ledgerCfg := repairsim.LedgerConfig{ + StartSlot: 10_000, + Slots: *slots, + FECsPerSlot: *fecSets, + EntriesPerSlot: *entries, + Seed: *seed, + ShredVersion: 1, + ReferenceTick: 63, + } + started := time.Now() + ledger, err := repairsim.GenerateLedger(ledgerCfg) + if err != nil { + fatalf("generate ledger: %v", err) + } + generationWall := time.Since(started) + result, err := repairsim.Run(ledger, network) + if err != nil { + fatalf("run simulation: %v", err) + } + cpu := *cpuLabel + if cpu == "" { + cpu = cpuModel() + } + report := report{ + Environment: environment{GoVersion: runtime.Version(), GOOS: runtime.GOOS, GOARCH: runtime.GOARCH, CPU: cpu}, + Ledger: ledger.Config, Network: network, LedgerGenerationWall: generationWall, Result: result, + } + encoded, err := json.MarshalIndent(report, "", " ") + if err != nil { + fatalf("marshal report: %v", err) + } + encoded = append(encoded, '\n') + if *output == "" { + _, _ = os.Stdout.Write(encoded) + return + } + if err := os.WriteFile(*output, encoded, 0o644); err != nil { + fatalf("write %s: %v", *output, err) + } +} + +func cpuModel() string { + if runtime.GOOS == "linux" { + if data, err := os.ReadFile("/proc/cpuinfo"); err == nil { + for _, line := range strings.Split(string(data), "\n") { + if key, value, ok := strings.Cut(line, ":"); ok && strings.TrimSpace(key) == "model name" { + return strings.TrimSpace(value) + } + } + } + } + if runtime.GOOS == "darwin" { + if out, err := exec.Command("sysctl", "-n", "machdep.cpu.brand_string").Output(); err == nil { + return strings.TrimSpace(string(out)) + } + } + return "unknown" +} + +func fatalf(format string, args ...any) { + _, _ = fmt.Fprintf(os.Stderr, format+"\n", args...) + os.Exit(1) +} diff --git a/config.example.toml b/config.example.toml index 7e3833087..4211c3871 100644 --- a/config.example.toml +++ b/config.example.toml @@ -328,6 +328,24 @@ name = "mithril" # Signature-verification workers (0 = GOMAXPROCS). tpu_sigverify_workers = 0 + # Optional minimum slot for NEW votes (inclusive), also --wait-to-vote-slot. + # Useful when rejoining after recovery. CLI overrides this setting. + # Zero adds no operator cutoff; the automatic startup cutoff and normal + # consensus checks still apply. A lower value cannot bypass those checks. + # Replay/repair continue while waiting. Previously recorded, authenticated + # votes can still be restored/rebroadcast under the existing recovery rules. + # This does not coordinate a cluster restart or wait for supermajority, + # and does not allow resetting a corrupt vote-history file. + wait_to_vote_slot = 0 + # Bounded cross-slot TPU queue. Zero keeps the 131,072-transaction default. + # Larger queues can prefill four busy leader slots, using additional memory. + tpu_max_buffered_transactions = 0 + + # Milliseconds reserved for local finalization and broadcast, not consensus + # finality. Zero keeps the conservative 75ms default. Tune from measured + # completion margins; this does not change the protocol slot deadline. + block_completion_reserve_ms = 0 + # ============================================================================ # [consensus] - Alpenglow Consensus # ============================================================================ @@ -493,6 +511,21 @@ name = "mithril" # Zstd decoder concurrency (defaults to NumCPU) # zstd_decoder_concurrency = 16 + # Streaming execution (Alpenglow, native turbine only): execute a block's + # entry batches while its remaining shreds arrive, on a speculative bank + # over the executed parent. The complete block stays authoritative — the + # executed prefix must be the block's own transactions (pointer identity) + # or the bank is discarded and the block executes whole. Off by default. + # streaming_execution = false + # Execution workers per streaming group (0 = min(txpar, 4)). + # streaming_workers = 0 + # Contiguous decoded batches to accumulate before a group executes + # (0 or 1 = execute as each batch arrives). + # streaming_min_group_batches = 0 + # Discard a speculative bank whose block has not completed after this + # many milliseconds (0 = 2000). + # streaming_max_open_ms = 0 + # Snapshot bootstrap I/O tuning. # These defaults deliberately avoid flooding a single NVMe with hundreds of # concurrent writes. Increase cautiously on very fast multi-disk systems. @@ -524,14 +557,6 @@ name = "mithril" # Number of borrowed accounts to preallocate in arena (0 to disable) borrowed_account_arena_size = 1024 - # ed25519 signature verification backend. - # auto - use the AVX-512 accelerated backend when the CPU has - # AVX512-IFMA (Zen 4/5, Ice Lake and newer), else portable - # r51 - force the accelerated backend; startup fails without AVX512-IFMA - # generic - force the portable pure-Go backend - # stdlib - use Go’s crypto/ed25519 implementation after the mandatory strict rejection checks. - sigverify_backend = "auto" - # Enable/disable pool allocator for slices use_pool = true @@ -574,6 +599,33 @@ name = "mithril" # Filename to write CPU profile (for offline analysis with go tool pprof) # cpu_profile_path = "/mnt/mithril-data/profiling/cpu.pprof" +# ============================================================================ +# [sigverify] - Transaction Signature Verification +# ============================================================================ + +[sigverify] + # ed25519 backend: auto selects AVX-512 IFMA when available, else portable. + # r51 requires AVX-512 IFMA; generic and stdlib force portable backends. + # Strict signature checks are always enabled. The older + # tuning.sigverify_backend key remains supported when this key is absent. + backend = "auto" + + # Shared Turbine transaction verification workers; 0 = min(2, GOMAXPROCS). + # Leaves execution and shred/consensus processing room to run concurrently. + # This does not change validator.tpu_sigverify_workers or replay's fallback pool. + workers = 0 + + # Signature lanes per group: 4 or 8 (0 also means 8). Transactions stay + # indivisible, so a multisignature transaction may exceed this target. + # Ready short groups run immediately; no timer waits for more shreds. + batch_target = 8 + + # Decode complete entry batches and verify while later shreds arrive. + # Set true to compare against completion-only verification. + # Early work reserves at most 8 slots and 64 MiB of encoded component bytes; + # decoded transactions and Go bookkeeping use additional heap memory. + disable_shred_overlap = false + # ============================================================================ # [debug] - Debug Logging # ============================================================================ diff --git a/docs/alpenglow_branch_engine.md b/docs/alpenglow_branch_engine.md index c155881fb..3171d0c1d 100644 --- a/docs/alpenglow_branch_engine.md +++ b/docs/alpenglow_branch_engine.md @@ -160,6 +160,31 @@ mixed (heterogeneous-client) or Mithril-only cluster identically: timeouts, vote signing/transmission, durable vote-history persistence, and standstill participation. +## Replay-observer diagnostics + +The observer retains certificate history for deduplication and match/mismatch +reporting. A separate bounded index contains only retained, block-bearing +certificates that have not yet been reconciled against replay. Reconciliation +removes an entry after either a match or mismatch; eviction removes it together +with the historical certificate. Hashless/skipped replay cannot reconcile a +block-bearing certificate. Pending counts and age/window statistics retain the +same semantics, but scan unresolved entries rather than completed history. + +This index is disposable, process-local diagnostic state. It neither authorizes +votes nor substitutes for verified certificates, the chain tracker's finality +checks, durable signing bounds, vote history, or checkpoint recovery. Those +checks and persistence contracts are unchanged. + +`BenchmarkObserverEmptyReplay` measures observer work for an empty block, with +or without four preceding skipped slots, against 4,096 retained certificates. +It covers 0, 32, and 4,096 unresolved entries. On Ryzen 9700X (GOMAXPROCS=8, +three 300 ms runs), median time for the four-skips-plus-empty case with 32 +unresolved entries was 686.4 µs before the index and 1.87 µs afterward. With +all 4,096 entries unresolved it was 380.5 → 159.3 µs. These are component +benchmarks; they exclude execution, certificate cryptography, network delivery, +and end-to-end FAST inclusion. Live comparisons must account for observer +history warming after a restart and different leader/skip patterns. + ## What this proves — and does not The certificate layer proves which block *data* the cluster settled on. In diff --git a/docs/certificate-processing-evidence.md b/docs/certificate-processing-evidence.md new file mode 100644 index 000000000..c7b520914 --- /dev/null +++ b/docs/certificate-processing-evidence.md @@ -0,0 +1,14 @@ +# Certificate Processing: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/72514abc5a2a98d2a2823fe92f0bfbbeedbcc9bc) +(tag `review-evidence-20260916-certificate-processing`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/72514abc5a2a98d2a2823fe92f0bfbbeedbcc9bc/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/certpool-offlock.md b/docs/certpool-offlock.md new file mode 100644 index 000000000..646d3e70b --- /dev/null +++ b/docs/certpool-offlock.md @@ -0,0 +1,78 @@ +# Incoming vote verification outside the pool lock + +The incoming vote pool previously held one mutex while checking BLS batches and folding signatures into aggregates. A 45-second trace on the live Zen 5 validator observed a 9.49 ms p95 acquisition wait and a 57.68 ms maximum across 34,580 normally completed AddVote calls. This blocked other incoming votes and readers; durable pruning could also wait for this lock while holding the consensus output mutex. + +The change retains one expensive BLS batch at a time using a separate verification mutex, while releasing the pool-state mutex around cryptography. One caller owns processing for each slot. Ordinary arrivals may join that slot's pending maps while verification runs, and the owner revisits state after those arrivals. Admission that needs authentication to resolve quota pressure or competing signatures waits instead of weakening bounds or first-packet-poisoning checks. + +Candidates remain pending until verification finishes, so in-flight votes count against admission limits and exact duplicates consume no additional space. Only selected candidates are removed on completion. Pruning and eviction may remove a slot during verification; pointer identity checks prevent stale work from resurrecting it or releasing accounting twice. If the epoch lookup, installed validator-set identity or shred version changes, the old slot state is retired instead of mixing bindings. Normal engine validator sets are immutable within an epoch. + +Each completed verified batch publishes promptly, outside both mutexes. New arrivals do not defer an authenticated quorum until the slot stops receiving votes. Reward-footer flushing waits for the active owner, drains relevant pending votes, and honors the publication barrier. Verified stake reads preserve freshness by waiting for that slot's owner. Snapshot counters and durable pruning do not wait for cryptography. + +Point reuse is included: aggregation uses already-verified signature points, failed batches subdivide parsed members, and the unused tally public-key aggregate is removed. Randomized coefficients, signature checks, stake thresholds, equivocation budgets and paired-vote disjointness remain. + +## Lock ownership + +`verifyAndFoldTallyWithLockReleased` requires the pool lock on entry and returns +with it held, including early exits. It releases the lock during crypto and +revalidates slot and validator bindings after reacquiring it. The caller owns +that slot's processing marker. `finishSlotAndUnlock` consumes lock ownership: +it releases the lock, emits certificates and returns unlocked. Callers must not +pair it with a deferred unlock. + +## Bounded multi-scalar aggregation + +With more than one Go execution thread, batches of at least 16 parsed votes +use gnark `MultiExp` for each weighted +public-key/signature sum. G1 and G2 run sequentially with `NbTasks: 1`; the +existing pool verification mutex still admits one expensive batch at a time. +Smaller batches keep the scalar loop because bucket setup costs more than it +saves, particularly during failed-batch subdivision and two-candidate checks. +Single-thread configurations also retain the scalar path: native contention +tests found that MultiExp task handoffs could increase certificate wall time +there, despite reducing arithmetic. + +Each member still receives a fresh, independent, nonzero coefficient sampled +from the full scalar field. Its public key and signature use the same +coefficient. Entropy/aggregation errors retain the individual-verification +fallback. Parsing, subgroup/infinity checks, failed-batch subdivision, +publication and stake accounting are unchanged. + +### Native staging validation + +AMD Ryzen 7 9700X, Go 1.26.4, GOMAXPROCS=2, Nice 15, two-core CPU quota, with +the normal validator workload still running. The baseline restores the scalar +implementation from the preceding certificate split head (`395e4566`) with +identical fixtures. Baseline/candidate/candidate/baseline runs were followed by +a final check after adding the single-thread safeguard. Full-fold ranges: + +| Votes in fold | Scalar baseline | Final implementation | +| --- | ---: | ---: | +| 32 | 7.385–7.386 ms | 4.238–4.344 ms | +| 64 | 14.217–14.280 ms | 6.800–7.135 ms | + +A controlled scheduling experiment runs 64 valid votes every 200 ms alongside +4,096 repeated transfer executions. With two Go execution threads, the final +alternating native comparison reduced certificate processing from +16.25–16.39 ms to 8.43–10.22 ms. Execution results were noisy: 13.80–14.38 ms +baseline versus 13.84–17.56 ms candidate. The earlier native comparison and the +local comparison showed no execution regression, but the slower final sample +is retained; the shared host does not establish absence of interference. +This workload excludes account commits, network delivery and vote persistence. + +The initial prototype's single-thread certificate latency could worsen while +execution improved slightly. The final implementation therefore keeps the +original scalar arithmetic whenever GOMAXPROCS is one. No worker-count or +verification-admission changes accompany this optimization. + +## Reproduce and validate + +Run `go test -race ./pkg/alpenglow` for concurrency, pending-budget, stale-binding, +invalid-share, equivocation and publication-barrier coverage. +Run `go test ./pkg/alpenglow -run '^$' -bench '^BenchmarkCertPool(WeightedPairing|FoldVerifiedBatch)$' -benchmem -benchtime=500ms -count=3` with the same fixture on each revision. + +The earlier concurrent 64-vote diagnostic reduced Snapshot p95 from +8.57–8.72 ms with point reuse alone to 0.008–0.038 ms with verification outside +the lock. This measures reader latency, not vote throughput. The diagnostic's +ns/op includes intentional sampling pauses. Historical component baselines, +live observations and full qualifications are preserved in the +[evidence archive](certificate-processing-evidence.md). diff --git a/docs/certpool-point-reuse.md b/docs/certpool-point-reuse.md new file mode 100644 index 000000000..cc2ac3a84 --- /dev/null +++ b/docs/certpool-point-reuse.md @@ -0,0 +1,18 @@ +# Reuse verified BLS signature points + +Incoming verification returns parsed, verified members and reuses their signature +points when folding the tally. Failed aggregate checks subdivide parsed members +instead of reparsing each subset. Individual verification returns the same member +representation. Installed validator sets retain their parsed public keys. + +The unused tally public-key sum is removed. Randomized verification still builds +its required weighted public-key sum with independent full-field coefficients. +Subgroup/infinity checks, invalid-share rejection, duplicate/equivocation checks, +stake accounting and paired-vote disjointness remain enforced. + +Point ownership and message association are covered by differential tests against +individual verification, including malformed signatures, invalid ranks and wrong +payloads. See [pool concurrency and aggregation](certpool-offlock.md) for the +current lock contract, combined implementation and benchmark method. +The isolated point-reuse experiment is preserved in the +[historical evidence](certificate-processing-evidence.md). diff --git a/docs/erasure_recovery_experiments.md b/docs/erasure_recovery_experiments.md new file mode 100644 index 000000000..5c47fe2fe --- /dev/null +++ b/docs/erasure_recovery_experiments.md @@ -0,0 +1,130 @@ +# Fixed-shape FEC recovery + +Production uses direct recovery only for exactly one missing data shard in a +32-data/32-coding FEC set with an available coding shard. All other availability +patterns keep the general decoder. Recovery still passes ordinary packet/root +validation before assembler admission. + +The deterministic production-path harness is described in [repair_sim.md](repair_sim.md). +Reduced-subset and all-coding plans remain reference benchmarks, not runtime +policies. Historical investigation notes are in the [evidence archive](fec-producer-evidence.md). + +## Matrix contract + +The experiment uses the systematic generator over `GF(256)/0x11d`: + +```text +V[x,j] = x^j +A = V[0:32,0:32] +G = V * A^-1 +G = [I_32; C] +``` + +For the fixed 32+32 shape, exhaustive scalar tests confirm all 1,024 entries: + +```text +C[r,c] = 0xa5 / (0x20 xor r xor c) +``` + +They also confirm `C*C=I`. Recovery tests remain differential against +`github.com/klauspost/reedsolomon`; the closed form is not the sole oracle. + +## Candidates + +### Near tip: direct one-data recovery + +For missing data position `m` and available coding position `r`: + +```text +D_m = C[r,m]^-1 * P_r + + sum(i != m, C[r,m]^-1 * C[r,i] * D_i) +``` + +This prepares one 32-source coefficient row and writes one destination. It does +not construct or invert a general 32x32 matrix. + +Production uses a process-wide table containing every missing-data and +coding-row combination. This removes per-call plan construction while keeping +the same equation. The table is exhaustively differential-tested against the +general decoder across all 32 x 32 combinations. + +### Catch up: reduced missing-data system + +For missing data columns `M` and selected coding rows `R`, substitute every +known data shard and solve: + +```text +B[i,j] = C[R[i],M[j]] +B * D_M = adjusted_coding_rows +``` + +The experiment uses the Cauchy closed form to construct `B^-1` in `O(m^2)`, +expands direct rows over 32 selected sources, and proves each row satisfies the +requested systematic generator row before byte processing. An independently +implemented Gauss-Jordan inverse is the setup fallback. + +The byte kernel is intentionally portable and uses `reedsolomon.LowLevel`. +This isolates algorithm and plan costs; it is not evidence that a portable +kernel will beat the dependency's generated AVX2/GFNI kernels on amd64. + +### Catch up edge: all coding rows + +When all data rows are missing and every coding row is present, `C*C=I` means +the existing optimized encoder can apply `C` to the coding rows and recover the +data directly. This is kept as a separate synthetic arm. It is simpler than a +general decoder, but a cached reduced-system plan may still have a faster byte +kernel; hardware decides between them. + +## Synthetic coverage + +The tests cover: + +- every missing-data position with every coding-row choice for the direct path; +- every pair of missing data positions; +- deterministic mixed patterns at 2, 4, 8, 16, 24, and 32 missing data shreds; +- exactly-threshold and one-below-threshold availability; +- changed availability between setup and execution; +- destination failure atomicity; +- coefficient mutation detection; +- Cauchy inverses against independent Gauss-Jordan inversion; +- recovered bytes against the existing general decoder. + +Run: + +```bash +go test ./pkg/turbine/internal/rsrecover +go test -run '^$' \ + -bench '^(BenchmarkRecoverOneData|BenchmarkRecoverDataSubset)$' \ + -benchmem -benchtime=2s -count=6 \ + ./pkg/turbine/internal/rsrecover +``` + +Benchmark result interpretation must keep these cases separate: + +- `prepare`: cold or changing erasure pattern; +- `execute`: prepared/repeated pattern; +- `prepare-and-execute`: first useful output for a new pattern; +- general cache off: existing decoder with changing pattern cost; +- general cache on: existing decoder after the inversion is cached. + +No production dispatch threshold should be chosen from an Apple benchmark. +Final crossover decisions require the pinned amd64 target and synthetic arrival +traces for progressing, stalled, and bursty slots. + +## Zen 5 production gate + +The direct one-data path was measured on a Ryzen 7 9700X with Go 1.26.4, +`GOMAXPROCS=1`, and one pinned physical core. Medians below are from seven +sequential one-second samples unless otherwise noted. + +| Benchmark | General path | Direct one-data path | Change | +| --- | ---: | ---: | ---: | +| one-missing `SlotAssembler` boundary | 10.73 us/FEC | 2.82 us/FEC | -73.7% (3.8x) | +| near-tip repair simulation | 3.0295 ms/op | 2.8659 ms/op | -5.40% | +| deep-mixed repair simulation | 4.0343 ms/op | 4.0240 ms/op | -0.26% | +| deep-sparse repair simulation | 10.8489 ms/op | 10.8327 ms/op | -0.15% | + +The production dispatch is intentionally narrow. The deep scenarios do not +enter it and remain effectively neutral, while the near-tip workload benefits +from repeated exactly-one-missing recoveries. The one-missing boundary also +dropped from 144 to 5 allocations per operation. diff --git a/docs/fec-producer-evidence.md b/docs/fec-producer-evidence.md new file mode 100644 index 000000000..68c57c7d2 --- /dev/null +++ b/docs/fec-producer-evidence.md @@ -0,0 +1,18 @@ +# FEC producer and recovery evidence + +The original #259 source, standalone producer benchmarks, raw samples and +recovery derivation are preserved at +[the historical snapshot](https://github.com/Overclock-Validator/mithril/tree/a3b16ebaaf803807ad04a7975f3eccf1c15649ea) +(tag `review-evidence-20260916-fec-producer`). + +The September 6 comparison used development head `7e4e8af1`, not today's #278 +base. Fifty thousand 1,232-byte legacy transactions across three slots took +734.052 → 282.296 ms on one pinned Zen 5 CPU. With both versions using the same +30,816-byte batch target, the result was 734.052 → 288.378 ms. This measures +serial producer work, excluding transaction execution, admission verification, +worker queue overlap, routing and network delivery. It is not a whole-validator +speedup or a benchmark of the rebased combined Turbine review. + +The rebase preserves newer slot-byte reservations and the asynchronous shred +worker. Fixtures explicitly use the legacy 1,232-byte limit rather than the +newer 4,096-byte transport maximum. Reusable benchmark code remains in source. diff --git a/docs/fec-recovery-authentication.md b/docs/fec-recovery-authentication.md new file mode 100644 index 000000000..a1b7b752b --- /dev/null +++ b/docs/fec-recovery-authentication.md @@ -0,0 +1,63 @@ +# Authenticating recovered FEC data + +Received shreds must pass leader-signature verification before entering the slot +assembler. Reed–Solomon reconstruction alone does not authenticate missing data: +a leader can sign a Merkle tree containing inconsistent data and coding shards. +Structural validation of recovered headers does not reject that case. + +Both the specialized one-missing-data decoder and the general decoder now require +`authenticateRecoveredFEC` to succeed before returning any recovered data. It +reconstructs missing coding shards too, builds the complete data/coding Merkle +tree, and compares its root with a received coding shred's signed root. Checking +only recovered-data proofs would not detect an inconsistent commitment to a +missing coding shard. Existing received packets are read-only throughout recovery. + +On success, recovered data receives complete Merkle proofs and the coding +template's chained root and, where applicable, retransmitter signature. The latter +is a hop signature copied from a received packet, not a reconstruction of a lost +relay's signature. On failure, no recovered data is published. The all-data-present +path does not reconstruct or authenticate another tree; it relies on ingress +verification of the received data. + +This follows [Agave's recovery algorithm](https://github.com/anza-xyz/alpenglow/blob/9f284c913f3c78b36179ae2461fa91286a616fb9/ledger/src/shred/merkle.rs#L670): +reconstruct all missing shards, validate recovered headers, compare the full root, +and populate proofs. Signature verification is not repeated after the root match. + +## Validation + +`TestRecoveredFECAuthentication` exercises one, three and all 32 missing data +shreds, chained and resigned packets, altered recovered bytes, and leader-signed +inconsistent parity both received and absent. Invalid sets return no recovered +shreds. Valid recovered packets verify with the leader's public key. + +`TestRecoveredFECAgaveSignedCapture` uses four FEC sets from the existing captured +Agave slot 1,752,420. It regenerates parity without signing a new root, checks the +original committed roots, and compares recovered authenticated bytes and proofs +with the capture. Transport repair nonces and relay-specific signatures are +handled separately. This is a captured-wire compatibility test, not a run of a +Rust recovery oracle. The older localnet recovery test also covers an unchained +1+17 layout, with its 2022 proofs replaced by current signed proofs. + +## Cost and reproduction + +Apple M4 Pro, `GOMAXPROCS=2`, five 200 ms samples, warmed encoder cache. The baseline +is `039ebd67`; times are medians for one FEC recovery, excluding ingress signature +verification and block execution. + +| Received / recovered shape | Before | Authenticated recovery | +|---|---:|---:| +| 31 data + 32 coding; recover one data | 2.47 µs | 30.17 µs | +| 31 data + 1 coding; recover one data and missing parity | — | 104.52 µs | +| 29 data + 3 coding; recover three data and missing parity | — | 117.93 µs | +| 32 coding; recover all data | — | 118.01 µs | + +The first case increases from 4,264 bytes / 5 allocations to 8,360 bytes / 6 +allocations per operation. Threshold-arrival cases also pay to reconstruct missing +parity and assemble its leaf bytes. These measurements establish the cost of the +correctness check, not a live FAST-score or whole-block performance result. + +``` +GOMAXPROCS=2 go test ./pkg/turbine -run '^$' \ + -bench 'Benchmark(RecoverFECOneMissingBoundary|AuthenticatedFECRecovery)$' \ + -benchmem -benchtime=200ms -count=5 +``` diff --git a/docs/leader-packing-evidence.md b/docs/leader-packing-evidence.md new file mode 100644 index 000000000..8a439d1eb --- /dev/null +++ b/docs/leader-packing-evidence.md @@ -0,0 +1,14 @@ +# Leader Packing: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/06ef067798c99947e8cc527450ad28430a9a7333) +(tag `review-evidence-20260916-leader-packing`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/06ef067798c99947e8cc527450ad28430a9a7333/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/leader_block_packing.md b/docs/leader_block_packing.md new file mode 100644 index 000000000..b30ad57a1 --- /dev/null +++ b/docs/leader_block_packing.md @@ -0,0 +1,105 @@ +# Leader block packing and synthetic load tests + +This change reduces work performed during a leader's available packing window. +The scheduler owns and decodes packet bytes once, and prepares static message +validation, instructions/account metadata, compute limits, message hash and cost +while transactions are queued. Each bank checks the immutable feature snapshot +before reuse. Account state, age, duplicates, strict fee-payer eligibility, rent, +execution and all resource budgets remain bank-dependent checks. + +Leader execution reuses borrowed-account scratch and skips detailed replay +stage timers. Missing-current-bank lookup errors defer base58 formatting until +used, avoiding wasted work before parent lookup. Entry Merkle hashing retains +only the root-building scratch, and max-heap removal avoids heap interface dispatch +while preserving priority/FIFO order. Both priority heaps now track entry +indexes so consumption, eviction and expiry remove every buffer reference. +Repeated rebuffering reuses the caller's intact transaction without accumulating +duplicate heap references. The existing slot-local skip scanning policy remains +unchanged, including selection of newly arrived higher-priority transactions. + +## Capacity and protocol limits + +Live banks use slot-dependent budgets with slot-time feature gates taking effect +in the epoch after activation. For the September 13 cluster's 200ms regime with +RaiseBlockLimitsTo100m, the budget was 50M block-cost units, 20M writable-account +cost units, 50MB allocated-data growth and 10MiB entry bytes. The entry packer +reserves 48 bytes for the ending tick. A 100M per-slot budget would be incorrect +in this regime. Active limits are logged when a leader bank opens. + +The readonly-pair fixture has one signature, one writable fee payer, two existing +readonly accounts, no instructions, and 198 wire bytes. Its observed actual cost +is 1,028 units: 720 signature, 300 write lock and eight loaded-account units. Its +program execution cost is zero. The theoretical cost-only ceiling is 48,638, +but upfront admission must fit the larger estimated loaded-data reservation: +the offline bank test includes 48,622 before rejecting the next transaction. +This workload is designed for signature/packing load, not application execution. + +## Reproduce locally + +All commands below are offline. They use deterministic test keys and in-memory +accounts; no RPC, faucet, funding or transaction submission occurs. + +```sh +# The 200,000-message, eight-payer, two-blockhash workload used to prefill four slots. +go test ./pkg/tpu/txfixture -run '^TestReadonlyPair200KDistinctMessages$' -count=1 + +# Fill a 50M-cost bank, reject the next tx, check fees, and round-trip all entries +# through actual shred generation and decoding, preserving transaction order/hash. +go test ./pkg/blockprod -run '^TestReadonlyPairBlockCapacityAndShredRoundTrip$' -count=1 + +# Whole-bank construction: one serial caller; pre-signed unique transactions. +GOMAXPROCS=8 go test ./pkg/blockprod -run '^$' \ + -bench '^BenchmarkReadonlyPair(FullBlock|PreparedFullBlock)$' -benchtime=3x -count=3 + +# More representative instruction workloads and smaller component microbenchmarks. +GOMAXPROCS=8 go test ./pkg/blockprod -run '^$' \ + -bench '^BenchmarkWorkingBank(Decoded|Prepared)?HotAccounts$' -benchtime=2s -count=3 +``` + +The whole-bank benchmark admits 48,622 transactions and includes execution, +account publication, entry batching/hash work and final entry flush. Signing and +bank setup are excluded. The prepared variant additionally does static +preparation before timing, modeling a queue ready before leadership. That work +is moved, not eliminated. Actual AccountsDB, signature verification, network, +consensus and the protocol deadline are outside this benchmark. The correctness +test's shred round-trip is also outside the timed benchmark. + +`txfixture.ReadonlyPairWire` provides the same ordered-pair construction as the +live experiment: 128×127 unique messages per payer/blockhash. Repeated ordinals +need a different payer or blockhash. The 200k test verifies uniqueness across +all messages and decodes/verifies representative signatures and phase boundaries. + +## Queue supply and completion reserve + +```toml +[validator] +tpu_max_buffered_transactions = 0 # default 131072 +block_completion_reserve_ms = 0 # default 75ms +``` + +The corresponding flags are `--tpu-max-buffered-transactions` and +`--leader-completion-reserve-ms`. The measured full-prefill trial used 262,144 +queue entries and a 60ms reserve. Those are opt-in tuning values; defaults stay +unchanged. A larger queue uses additional memory for owned wire, decoded and +prepared objects. `BenchmarkReadonlyPairPreparationMemory` measured 728 allocated +bytes per preparation (9 allocations) on Go 1.26.4 arm64 for the 198-byte fixture. +That is 91 MiB of allocation volume for 131,072 preparations, or 182 MiB for +262,144, in addition to wire/decoded transactions and queue indexes. Allocation +volume includes temporary preparation storage: it is not retained heap or RSS. +More accounts/instructions increase the footprint; these are not worst-case caps. +Reproduce with `go test ./pkg/blockprod -run '^$' -bench '^BenchmarkReadonlyPairPreparationMemory$' -benchmem`. +A shorter reserve needs measured local finalization/broadcast +margin and does not change the protocol deadline. Shifting completion also +shifts later bank start times, so it does not add the same packing time to all +four blocks. + +The live results showed why total supply and timing matter: 120k transactions +cannot fill four approximately 48.6k blocks. An 80k refill competed with ongoing +work. Preloading 200k into a larger queue improved the observed four-block total, +while the first block still had less usable time and later banks awaited local +replay/adoption. These observations do not isolate a single CPU bottleneck. + +[Measured results, exact baselines and evidence](https://github.com/Overclock-Validator/mithril/blob/06ef067798c99947e8cc527450ad28430a9a7333/docs/results/leader-block-packing/2026-09-13/README.md) +include both the successful near-limit block and the still-underfilled four-slot +window. The archived live helper is historical experiment source with explicit +cluster/identity/path constants; the offline fixture is the portable reproduction. diff --git a/docs/out-of-order-entry-prefetch.md b/docs/out-of-order-entry-prefetch.md new file mode 100644 index 000000000..9645b6d3d --- /dev/null +++ b/docs/out-of-order-entry-prefetch.md @@ -0,0 +1,28 @@ +# Prepare complete entry batches despite earlier shred gaps + +Live tracing found five large external blocks whose final assembly-to-ready time was 20–39 ms. Most fallback verification work was already available more than 20 ms before full assembly. For one 33,760-transaction block, 6,506 transactions in later complete batches waited 44–116 ms for discovery behind an earlier missing shred. + +The old discovery cursor stopped at the first missing data index. The new bounded bitmap index discovers each complete DATA_COMPLETE range independently. Receiving a data shred or recovering one through FEC can release its containing batch and, when it supplies a boundary, the next batch. A batch still needs every data shred and its preceding boundary, unless it begins at index zero. Results are queued in discovery order and final assembly restores wire order using the existing exact range and byte-identity checks. + +The index uses about 24 KiB per retained slot and is allocated only with streaming preparation enabled. Successor/predecessor queries have bounded cost even with reverse or adversarial arrival order. Existing worker counts, verifier batching, retained-byte limits, cancellation, generation ownership, final block checks and signature validation remain unchanged. + +Regression coverage includes a delayed earlier shred, delayed preceding boundary, FEC-recovered boundary, disabled preparation, randomized arrival against a reference oracle, duplicate discovery, index word/group/slot boundaries, and the existing final-validation and cancellation tests. Native Turbine/replay race tests and Turbine vet passed. + +## Controlled Zen 5 benchmark + +AMD Ryzen 7 9700X, Go 1.26.4, Narya r51, GOMAXPROCS=8, two transaction verification workers, target batch size eight. 33,760 generated signed transactions arrive as complete component bursts across 200 ms. One data shred in a component three quarters through the block is withheld until after the footer. Both versions run with identical fixtures in baseline/candidate/candidate/baseline order, three iterations per case in each run, while the validator remains running. Ranges below are the two run medians, not a confidence interval. + +| Wire transaction size | Prior assembly → ready | New assembly → ready | +| --- | ---: | ---: | +| 228 bytes, delayed shred | 21.43–22.77 ms | 2.82–2.83 ms | +| 1,232 bytes, delayed shred | 35.46–36.34 ms | 7.04–8.08 ms | +| 228 bytes, ordered | 2.50–2.56 ms | 2.40–2.52 ms | +| 1,232 bytes, ordered | 6.91–7.67 ms | 6.87–7.89 ms | + +The delayed-shred case now verifies roughly 33.5–33.7k transactions before assembly, compared with 25.3–25.6k previously. The benchmark asserts every retained signature is verified exactly once and block metadata remains correct. It includes assembly, decoding, final validation and transaction verification; it excludes network I/O, replay execution and actual FAST-certificate inclusion. This modeled case is not an end-to-end validator speedup or a prediction of overall FAST percentage. + +Reproduce with `MITHRIL_SIGVERIFY_FLOW_BACKEND=r51 GOMAXPROCS=8 go test ./pkg/turbine -run '^$' -bench '^BenchmarkEntryPrefetchGapArrival$' -benchtime=3x` on each implementation, copying the same benchmark file to the baseline. + +Historical live trials and their limitations are in the +[archived evidence](streaming-preparation-evidence.md). Component results do not establish +a sustained FAST-inclusion improvement. diff --git a/docs/repair_sim.md b/docs/repair_sim.md new file mode 100644 index 000000000..b0240f015 --- /dev/null +++ b/docs/repair_sim.md @@ -0,0 +1,153 @@ +# Deterministic repair simulation + +`cmd/repair-sim` is a single-process harness for measuring the local path from +an incomplete slot to a block that is available to replay. It exists because +near-tip repair and deep catch-up optimize different outcomes: + +- near the tip, latency of the first replay-blocking slot matters; +- during catch-up, sustained useful data and completed slots per second matter. + +The first implementation deliberately stops before UDP and transaction +execution. It establishes a deterministic, correctness-checked baseline before +network realism or alternative scheduling policies are introduced. + +## What is real and what is simulated + +| Stage | Implementation | +| --- | --- | +| block-component serialization | production `turbine.MarshalBlockComponent` | +| 32+32 FEC generation and Merkle signing | production `turbine.Shredder` | +| missing-shred selection | production `SlotAssembler.RepairRequests` | +| packet parsing and Merkle/signature validation | production `ParseShred` and `ShredSignatureVerifier` (the receiver's per-root cache) | +| verified-shred insertion | production `ShredSpool` | +| threshold detection and Reed-Solomon recovery | production `SlotAssembler.AddShredFrom` | +| component decode and transaction-signature gate | production slot completion path | +| remote peer, latency, jitter, loss, duplication, bandwidth | deterministic in-process simulator | +| replay notification | block emission is recorded as “offered to replay” | +| transaction execution | not run in this version | + +Synthetic entries are valid Alpenglow entry-batch components but contain no +transactions. This isolates shred/FEC/repair/storage costs; it is not a replay +execution benchmark. + +## Scenarios + +### Near tip + +`near-loss` alternates two useful patterns across FEC sets: + +- 30 data + 1 coding shred: one repair response crosses the threshold and + reconstructs the other missing data shred; +- 31 data shreds: the one missing data shred must be fetched directly. + +Selected omitted shreds can also arrive through the simulated live path while +a repair response is outstanding. This measures cancellation/late-response +behavior without changing production scheduling. + +Disabling repair stops request scheduling but still delivers natural late live +shreds. The run ends when all slots complete or those live arrivals are +exhausted; incomplete slots are reported without a repair-stall error. + +### Deep catch-up + +`mixed` begins every FEC set with 16 data + 15 coding shreds. One fetched data +shred crosses the threshold and reconstructs the remaining 15. + +`sparse` begins every FEC set with two data shreds and no coding layout. The +ordinary repair interface serves data shreds only, so nearly all missing data +must arrive over the simulated network. Comparing `mixed` with `sparse` +quantifies the network work avoided by already-held coding shreds. + +## Commands + +```sh +go test ./pkg/turbine/repairsim + +go run ./cmd/repair-sim \ + -scenario=near-tip \ + -slots=200 \ + -fec-sets=4 \ + -seed=1 \ + -cpu-label='Ryzen 7 9700X' \ + -output=/tmp/repair-near.json + +go run ./cmd/repair-sim \ + -scenario=deep-catchup \ + -availability=mixed \ + -slots=1000 \ + -fec-sets=4 \ + -seed=1 \ + -repair-latency=20ms \ + -repair-bandwidth=104857600 \ + -output=/tmp/repair-deep-mixed.json + +go run ./cmd/repair-sim \ + -scenario=deep-catchup \ + -availability=sparse \ + -slots=1000 \ + -fec-sets=4 \ + -seed=1 \ + -repair-latency=20ms \ + -repair-bandwidth=104857600 \ + -output=/tmp/repair-deep-sparse.json + +go test ./pkg/turbine/repairsim \ + -run '^$' -bench '^BenchmarkScenarios$' -benchmem -count=5 +``` + +Logical trace timestamps are deterministic. `wall_elapsed_ns`, allocations, +and `stage_cpu_ns` are actual local measurements and therefore are not expected +to be byte-identical across runs. + +## Correctness gates + +The current tests require: + +- exact requested FEC-set counts from authentic generated shreds; +- Merkle/signature validity for every canonical packet; +- no completed slot when loss is present and repair is disabled; +- complete, canonical entry streams after threshold recovery; +- a complete `ShredSpool` journal record before replay admission; +- rejection and retry of a corrupted repair response; +- deterministic logical traces for identical seeds; +- late repair responses to leave a completed block unchanged. + +The Turbine package also retains focused byte-for-byte Reed-Solomon recovery +tests. The simulator verifies the stronger end-to-end consequence: recovered +shreds must decode to the exact canonical entry sequence and pass the normal +completion gates. + +## Generator compatibility finding + +Building this harness exposed a multi-FEC component bug: the local generator +set `DATA_COMPLETE_SHRED` at every FEC boundary. The decoder correctly treats +that flag as the end of one serialized component, so a component spanning more +than one FEC set was truncated and failed to decode. The generator now follows +Agave's ordering: construct every FEC set, then mark only the final data shred +of the component complete (or last-in-slot). A regression test round-trips one +1,300-entry component across multiple FEC sets. + +## Future mode selection + +No production mode switch is added here. A later policy experiment should use +both replay distance and observed Turbine usefulness, with hysteresis: + +- stay in near-tip mode while the replay gap is small and Turbine supplies a + high fraction of useful shreds before repair deadlines; +- enter catch-up mode only when the replay-blocking gap is sustained and live + Turbine delivery is insufficient to approach FEC thresholds; +- return to near-tip mode only after both the gap and repair backlog fall below + lower thresholds. + +That signal is preferable to slot distance alone: a node may be numerically +close to the tip while receiving too few live shreds, or far behind while its +local spool already holds most FEC thresholds. + +## Next steps + +1. Add shallow catch-up and whole-block-pressure configurations. +2. Add a loopback-UDP transport without replacing the deterministic mode. +3. Expose internal FEC start/finish timing through opt-in instrumentation. +4. Run replay execution against a reusable synthetic bank fixture. +5. Compare ordinary requests with explicit test-only threshold-acquisition and + earliest-blocked-slot policies. diff --git a/docs/reserved-vote-history.md b/docs/reserved-vote-history.md new file mode 100644 index 000000000..bd94e2e57 --- /dev/null +++ b/docs/reserved-vote-history.md @@ -0,0 +1,162 @@ +# Voting persistence and crash recovery + +## Intended guarantee + +A crash must not lead to conflicting externally published votes or reserved-mode +leader actions because the validator forgot its earlier local decisions. The design permits +loss of recent detailed history and sacrifices voting availability when its +completeness is uncertain. It does not promise immediate restart voting, that +every signed vote reaches disk, or recovery from rolled-back safety files. +Normal anti-equivocation, execution, parent and validator-binding checks remain +necessary; this is a persistence contract, not a proof of the entire protocol. + +The failure model includes process termination and host/power failure, provided +successful file and directory syncs survive, the current safety files are +preserved, and only one fenced owner uses the signing identity. Software tests +exercise the recovery decisions; they do not qualify actual storage against +power loss. Valid signatures on saved files establish integrity, not freshness. + +## Two different publication guarantees + +| Mode | Required before a vote can escape to the pool/network | What a restart may trust | +| --- | --- | --- | +| Default synchronous history | Exact validated history is written, file-synced, renamed and directory-synced before local pool admission or network enqueue. | Retained exact decisions and their rooted boundary, subject to normal restoration checks. | +| Opt-in reserved history | The vote's slot is covered by an acknowledged durable reservation before signing. Its exact history snapshot is prepared and queued before publication, without waiting for per-vote I/O. | The startup reservation bound, unless a separately validated clean-history seal proves exact retained history. | + +In synchronous mode, BLS bytes may be computed privately in RAM before history +is synced. The guarantee is **persist before publication**, including local +pool admission because it can publish a certificate. In reserved mode, a +successful queue submission, background rename, or `written` counter is **not** +a durable acknowledgement of that vote. Replacing a whole file atomically is +not the same as making its bytes/directory entry survive power loss. + +Both modes retain complete decisions in memory while running and prune them +only through the normal verified-root rules. The synchronous history guarantee +covers votes; it is not a complete journal of produced leader blocks. The +additional leader reservation barrier applies only in reserved mode. + +## Reserved-mode restart rule + +Let **H** be the reservation's `Through` value loaded at startup, **F** the +verified finality/checkpoint floor, and **S** a proposed signing slot. H bounds +what the previous process *might* have signed; it is not its last actual vote. + +Without a validated clean seal, every vote type and historical local-vote +restoration must obey all of these conditions: + +- **S > H**: never re-sign in the uncertain range during this run, even when + an older history file contains that particular vote. +- **F >= H**: do not sign above the range until verified finality/checkpoint + state has reached its end. F equal to H is sufficient; S equal to H is not. +- **S <= the current acknowledged reservation**, plus all ordinary protocol, + live-joining and configured minimum-slot checks. + +The startup recovery bound stays fixed during the run. Background renewal may +raise the current signing allowance; it does not move the recovery target. +Newly received blocks, elapsed wall time, an RPC tip, replay progress alone, and +`--wait-to-vote-slot` cannot substitute for verified finality/checkpoint state. +The recovery wait can be indefinite if the cluster halts below H. + +For example, suppose votes through 1,015 escaped, detailed history survives only +through 1,012, and the durable reservation is 1,032. After an unclean restart, +slots <= 1,032 remain forbidden. Slot 1,033 is also forbidden while F < 1,032. +Once F >= 1,032, it can pass the recovery gate only after an acknowledged grant +covers 1,033 and all normal voting checks pass. Seeing a new block after restart +does not by itself meet these conditions. + +## Clean shutdown is a durable protocol + +1. Stop/join the voter and leader producer. Halt/join the reservation worker; + drain/join the history writer so an older rename cannot overwrite the seal. +2. Require no latched safety fault or history-write failure, no unresolved + reservation-write uncertainty, and verified recovery through the startup H. +3. Sync exact retained history, including the verified rooted boundary. +4. Sync a reservation record containing the digest of that exact history. + +A successful process exit, a signal handler running, or an attempted final save +is not sufficient. Startup must load and validate the history and reservation, +match the clean digest, and **durably consume the clean marker with a dirty +successor before new vote/leader signing or new history decisions**. A later crash uses +the reservation again, even if the old history still looks valid. + +| Restart state | Vote recovery | Leader recovery | +| --- | --- | --- | +| Valid history and reservation, no matching clean seal | Enforce S > H and F >= H, including restoration. | Enforce S > H and F >= H. | +| Valid matching clean seal, successfully consumed | Resume using exact retained voting decisions and ordinary checks; no additional vote quarantine through H. | Still enforce S > H and F >= H: vote history does not enumerate every block that may have been signed. | +| Missing, corrupt, unreadable or incompatible enrolled state | Refuse automatic reset/startup; do not infer safety from an RPC tip. | Same refusal. | + +Errors during sealing do not authorize treating the session as clean; startup +must validate whichever durable record survived. A failed reservation write +may have reached storage despite its error. Running signers retain only the +previous acknowledged allowance; a later monotonic successful write can resolve +that uncertainty. An unresolved error prevents deliberately sealing clean. + +## Lifetime, storage and enrollment + +The reservation grants up to 32 slots beyond the requested slot and renews when +16 or fewer remain. Only successful file-and-directory sync acknowledgement +publishes new permission. Exhaustion pauses signing while replay/verification +continue. Repeated restarts without new grants do not advance the bound. + +Detailed history uses one ordered writer with one in-flight and at most one +newer pending complete snapshot. New complete snapshots may supersede unwritten +ones. Validation/encoding/signing of the snapshot remain on the voter goroutine; +only file I/O is asynchronous. Writer errors latch a safety fault and stop +voting. An already in-flight vote is still covered by its durable reservation. + +Preserve both `vote_history-.mithril.json` and +`vote_reservation-.mithril.json` independently of AccountsDB snapshots. +The checkpoint encoding cache is only an encoding optimization; it is not the +vote journal or a replacement for the reservation. Rolling back AccountsDB or +an application binary must not roll back either signing-safety file. + +The directory lock only excludes concurrent owners of the same history path. +It cannot fence copies of the identity on other hosts/paths. Signed records and +generation numbers cannot detect an operator restoring an old valid pair of +safety files. Media loss, stale safety-file restoration, dishonest sync behavior +and compromised/copied signing keys are outside this automatic recovery contract. + +Use `--reserved-vote-history` to opt in; default persistence remains synchronous. +First enrollment additionally requires `--initialize-vote-reservation`, exclusive +identity ownership and complete synchronous history from the stopped previous +writer. An empty directory is appropriate only for a previously unused identity. +The software cannot distinguish that case from deleting both files for an old +identity; initialization is not a safe disaster-recovery reset. Remove the +initialization flag afterward. These are CLI flags, not TOML settings. + +Enrolled history uses version 2, rejected by older binaries that do not enforce +the reservation. Missing/corrupt enrolled state or a changed identity, vote +account, authorized voter, genesis or shred version must not be automatically +re-enrolled. Disabling the flag or deleting files is not a supported downgrade. +Transport binding/validation runs before the clean marker is consumed. + +## Live admission is separate from restart recovery + +The live vote admission floor uses retained consensus-pool state and the history +root, not the newest finalization certificate. Replay can still contribute a +notarization within the bounded 16-slot retained tail before later certificates +are collected; finalized retained slots can also receive skips. Durable-root +pruning is ordered behind completed replay events on the voter. These admission +changes also apply to synchronous mode. They do not weaken reserved-mode F >= H. +`--wait-to-vote-slot N` adds an inclusive minimum; it cannot bypass recovery, +execution, retained-root or parent checks. + +## Evidence and limits + +Existing tests map the recovery contract to these cases: + +| Contract | Tests | +| --- | --- | +| Lost valid history suffix; fixed bound across repeated crashes | `TestReservedVotingLostHistorySuffixAndRepeatedCrash` | +| All five vote types, restoration and leader gates at H/H+1 | `TestReservedVotingEverySignatureTypeAtBound` | +| Clean-marker consumption and stricter leader restart | `TestReservedVotingCleanMarkerConsumedBeforeSigning` | +| Digest mismatch, missing/corrupt/domain-mismatched state | `TestReservedCleanDigestMismatchUsesCrashRecovery`, `TestReservedVotingRejectsMissingCorruptOrWrongDomain` | +| Pending/uncertain sync cannot authorize signing | `TestSigningReservationUnacknowledgedSyncCannotAuthorize`, `TestSigningReservationUncertainWriteSurvivesRestart` | +| Writer drain, failure and lost pending snapshots | `TestAsyncHistoryBlockedWriteDoesNotDelayVotesAndCleanCloseDrains`, `TestAsyncHistoryFailureStopsVoterAndPreventsCleanMarker`, `TestAsyncHistoryProcessCrashLosesPendingSnapshots` | + +The subprocess-kill test exercises lost in-flight/pending application snapshots; +it does not power-cycle a host or prove filesystem durability. Fault-injection +and older-history fixtures check the algorithm's decisions under the stated +storage contract. This is not a formal consensus proof or mainnet storage +qualification. Persistence benchmarks likewise do not establish replay or FAST +performance. diff --git a/docs/rewards-unwind-retirement.md b/docs/rewards-unwind-retirement.md new file mode 100644 index 000000000..edf207be8 --- /dev/null +++ b/docs/rewards-unwind-retirement.md @@ -0,0 +1,58 @@ +# Retiring durable rewards bookkeeping + +A completed partitioned-rewards distribution used to leave its in-memory +descriptor alive for the rest of the replay attempt. The fork-switch guard +rejects any such descriptor because account-overlay unwind cannot restore the +consumed spool or its distribution counters. This is necessary while completion +is speculative, but unnecessarily forces checkpoint replay after completion +has become durable. + +Replay now observes the inactive EpochRewards sysvar in a successfully executed +bank's immutable snapshot, with zero partitions remaining. It remembers that +bank's slot and the exact distribution descriptor. Only applying a successful +durable fold through that slot retires the descriptor. Later bank observations +do not move the completion slot forward. A new descriptor/epoch invalidates the +old evidence; missing sysvars or unknown completion retain the old fallback. + +## Safety and recovery contract + +- Completion in memory, certificate finality, and submitting a fold do not + authorize retirement. Failed folds leave the durable watermark unchanged. +- Active distribution and completed-but-not-durable distribution retain the + existing rewards guard. No spool reconstruction or rewards rollback is added. +- After retirement, in-memory switches still require the existing epoch, + vote/stake-cache, parent-context, sysvar and transaction-status checks. + Switches at/below the durable watermark still require durable recovery. +- Completion evidence is replay-thread-owned and process-local. It does not + change checkpoint formats, signing reservations, persisted vote history, + clean-shutdown rules or restart authorization. Restart retains the existing + persisted EpochRewards validation. No extra file or disk sync is introduced. + +## Incident motivating the change + +On Zen 5, distribution completed at slot 3,942,001. At a later parent-linked +switch, the durable checkpoint was already 3,944,067; child 3,944,076 selected +parent 3,944,073, abandoning the suffix from 3,944,074. The remaining descriptor +forced the rewards-window fallback even though completion was below the root. +Checkpoint recovery re-fetched previously received blocks, with logged waits +of 2.739 seconds and 0.967 seconds. A buffered 665-transaction block waited +3,613.510 ms for replay admission and then executed in 7.520 ms. + +These are incident observations, not a before/after benchmark or a measurement +of checkpoint encoding/fsync time. Thirteen observed FAST aggregates omitted +our vote during the recovery interval; that does not prove absence from every +FAST aggregate or a single cause for all thirteen omissions. No live latency +improvement is established until a comparable switch exercises the new path. + +## Validation + +`rewards_retirement_test.go` covers active/missing bank state, unknown completion, +the exact durable boundary, later-bank observations, generation changes, failed +and successful folds, and an exact-parent unwind after retirement (including +account values, resume state and immutable rewards sysvars). Existing unwind +tests still require fallback for zero-remaining bookkeeping without retirement, +cross-epoch switches, dirty vote/stake caches and invalid parent snapshots. + +Full replay/rewards race suites passed locally and in the combined native +build; native node recovery/checkpoint race tests, vet and validator build also +passed. These are software tests, not mainnet power-loss qualification. diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md new file mode 100644 index 000000000..c399f4da5 --- /dev/null +++ b/docs/sbpf-interpreter-benchmarks.md @@ -0,0 +1,71 @@ +# Execution performance validation + +## Program workloads + +The program harness measures instruction setup, serialization, execution and +publication with a warm program cache and fresh account data per invocation. +It checks return values, account updates, CPI and CU consumption. Loader-only +benchmarks separately measure VM execution and program loading. + +Set `MITHRIL_PROGRAM_BENCH_DIR` to a directory containing `rotation_compute.so` +and `token2022.so` to enable those external fixtures. Without it, the in-repository +BPF/CPI fixtures still run. Pinned input SHA-256 values: + +- Arithmetic ELF: `db7c55d6563c879e35dfe2b24edb0fe0515d5a5ae627fe00e3e247001c786441`. + Source: `ag-transaction-bench` at `7e5a263fa5a1c72088f191daf5b7c5d2484c997c`, + `transaction-bench/program/src/rotation_compute.c`. +- Token-2022 ELF: `a794161408080f690dac00832f45b3c3e2b71f1339586667ad1f979cf91d5b68`. + Public Alpenglow program `TokenzQdBNbLqP5VEhdkAS6EPFLC1PHnBqCXEpPxuEb`, + fetched at RPC context slot 4,231,444, program-data account + `DoU57AYuPFu2QU514RktNPG22QhApEjnKxnBcu4BHDTY`. Strip its 45-byte upgradeable + loader metadata before saving the ELF. Verify the hash; do not silently replace + it with a later deployment. + +``` +MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ + go test ./pkg/sealevel -run '^TestProgramWorkloadResults$' -v +MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ + go test ./pkg/sealevel -run '^$' -bench '^BenchmarkProgramWorkloads$' \ + -benchtime=1s -count=5 -benchmem +``` + +## Correctness and comparison boundaries + +The generated-program harness compares return values, errors, CU usage and memory +for 100,000 generated programs, evenly divided across SBPF v0–v3. The generator +uses v2-specific memory, arithmetic and constant-loading encodings; the dump +reports verifier rejection or execution results, and logs accepted counts per version. +Run the same harness on both trees: set `SBPF_DIFF_OUT` separately on the reference +and candidate and compare the files. `SBPF_CHECK_POOL_ZERO=1` additionally checks +the candidate’s clear-on-return pool invariant; older references may clear on +acquisition instead. ARSH and verifier semantics changes are excluded from this +performance work. + +For replay comparisons, use fresh isolated AccountsDBs from the same snapshots, +identical transaction parallelism, and the same slot interval. Compare normalized +per-slot bank hashes and slot sets before interpreting timings. Compare exact +`ProcessBlock` wall-clock timers, not summed instruction/worker timers. Alternate +run order and retain raw outputs plus commit IDs outside the merge diff. + +Record tested commit IDs, hardware, Go version, affinity and parallelism with results. +Single-core shared-host results do not establish multicore contention or live FAST +inclusion gains. + +## Memory syscalls and LtHash + +Memory syscalls retain CU charges, source-before-destination error order, +zero-length behavior, memcpy overlap rejection and memmove overlap support. +Tests cover copy-on-write/growing regions and differential memory/CU results. + +LtHash uses AVX2 only when supported by both CPU and OS; other architectures and +`-tags purego` use portable loops. The vector path preserves 16-bit wraparound and +in-place operand aliasing. Randomized, unaligned, inverse and fallback tests cover +both paths. Component speedups are not block-latency speedups. + +```sh +go test ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader +go test -tags purego ./pkg/lthash +go test -race ./pkg/sealevel -run 'TestSyscallMem|TestMemoryCopyDifferential|TestProgramWorkloadResults' +SBPF_DIFF_OUT=/tmp/candidate-diff.txt SBPF_CHECK_POOL_ZERO=1 go test ./pkg/sbpf -run TestDifferentialDump -count=1 +go test ./pkg/lthash -run '^$' -bench BenchmarkMix -benchmem -count=5 +``` diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md new file mode 100644 index 000000000..4045b2bac --- /dev/null +++ b/docs/sha256-syscall.md @@ -0,0 +1,44 @@ +# SHA-256 syscall validation + +The syscall decodes the translated slice descriptors directly and writes the +final digest into the translated output buffer. It retains streaming SHA-256, +slice order, memory translations, CU charges and validation order. Output is +written only after all inputs have been read, preserving overlapping-buffer +behavior. + +The frozen reference implementation in the test harness supports differential +checks and before/after benchmarks. Both variants use the same VM, including +its contiguous-region bounds-overflow fix. + +Differential tests compare hashes, return/error values, remaining CU and all +input/output memory over valid inputs, invalid descriptors/addresses, depleted +budgets and output aliasing. The existing SHA program fixture also executes. + +## Benchmarks + +`BenchmarkSha256Syscall` measures the syscall through a real interpreter's memory +translation and CU meter, with VM creation outside the timed region. It covers +empty, single-slice, multiple-slice and larger inputs; it excludes instruction +dispatch and transaction execution. + +`BenchmarkSha256CapturedLoop` uses a captured SBF v0 hash loop with 1,000 +iterations and zero initial state. It retains descriptor setup, stack accesses, +digest copying, counter updates and branching. Its test checks the result against +a Go hash chain and checks equal CU consumption for both syscall implementations. +The captured ELF hash and extracted instruction range are recorded in the test. +Transaction loading, CPI and the rest of the original program are excluded. + +The dispatch-only control omits hashing, translations and syscall CU charging; +it is an overhead diagnostic, not a valid execution implementation. The raw Go +hash chain provides another comparison outside the VM. Neither control can +establish a full-block speedup. + +```sh +go test ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256|TestSha256CapturedLoop' +go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256(Syscall|CapturedLoop)$' -benchtime=1s -count=5 -benchmem +``` + +Alternate baseline/candidate run order and record commit IDs, Go version, +hardware and CPU affinity. Report component timings separately from full-program +and replay timings; hardware SHA acceleration also affects these results. diff --git a/docs/shred-spool-completion.md b/docs/shred-spool-completion.md new file mode 100644 index 000000000..b726265f3 --- /dev/null +++ b/docs/shred-spool-completion.md @@ -0,0 +1,85 @@ +# Shred-spool completion publication + +`MarkComplete` runs on the block-delivery path. Previously it wrote a 16-byte +`complete.idx` record while holding the spool mutex. This write uses no fsync, +but ordinary filesystem writes can still wait. Earlier live observations included +32–40 ms completion-to-delivery delays; one separate detailed trace localized a +37.8 ms pause to the journal write on an already-adopted own-leader block. That +own-leader example did not delay its vote. Do not equate all completion-to-delivery +stalls with journal I/O without the finer trace. + +Completion publication now updates the in-memory map and makes a nonblocking +submission to one writer with a 256-record queue. The writer never takes the +spool mutex. A full queue drops only the persistent completion hint; in-memory +completeness remains available. This bounds pending memory and prevents storage +backpressure from directly reaching completion publication. No worker-count or +validator configuration changes are required. + +## Recovery and ownership contract + +The spool is a disposable verified-shred cache, not vote history, account state, +or a durable checkpoint. Completion hints can be lost on an unexpected stop or +queue overflow. Recovery then reassembles/re-repairs; a hint never replaces the +assembler's coverage and block validation. The packet checksums and journal/file +formats are unchanged. There is no new fsync or power-loss durability guarantee. + +Deletion and corrupt-tail truncation are different: they must not race an older +queued completion. Their tombstone goes through the same ordered writer, and the +caller waits for it **before changing the slot file**. Thus an older queued hint +cannot be written after the tombstone and resurrect completeness for a replacement +partial file. These uncommon operations can still wait for storage while holding +the spool mutex; this change does not remove every source of spool contention. + +After a short/error write, the worker stops appending hints and truncates the +journal to zero. This avoids appending behind a partial record and removes old +completion hints before an invalidation is acknowledged. If truncation also fails, +the invalidation fails and the slot-file mutation is refused; a later attempt can +retry. Losing all cached completion hints is an acceptable repair-cost fallback. + +`Close` excludes further mutations, flushes packet buffers, retries current +completion hints if the queue overflowed, and drains/closes the journal worker +before returning. The next opener therefore preserves the existing clean-handoff +contract when storage succeeds. A stuck disk can still delay shutdown. The worker +must not outlive ownership of the spool directory. No voting-resume, signing-bound, +checkpoint-coverage or consensus-safety rule changes. + +## Validation and measurements + +Tests block the writer and overflow the queue while asserting that completions +remain available, then verify clean-close recovery. A replacement-file test keeps +the old completion write blocked and verifies that replacement cannot proceed +before its tombstone. Fault tests inject partial writes and failed truncation, +verify refusal to mutate the slot, then retry and check that no stale hint returns +on restart. Existing checksum/torn-tail, retention, handoff, receiver shutdown and +invalid-block tests remain covered. + +Native full turbine/blockstream race suites, vet and the combined validator build +passed on the Ryzen 9700X (Zen 5), Go 1.26.4. The test process used GOMAXPROCS=2, +Nice=19 and a 200% CPU quota on the active validator host. + +Run the identical `BenchmarkShredSpoolMarkComplete` file on both source revisions. +Each sample opens a fresh spool, appends a packet outside the timer, times one +completion, then closes/drains outside the timer. Thus no already-complete dedupe +or queue-overflow drop is measured. Three runs of 300 iterations: + +| Component | Before | Candidate | +|---|---|---| +| Median run p50 | 2.805 µs | 0.170 µs | +| Median run p99 | 6.201 µs | 0.581 µs | + +This measures ordinary storage, not injected tail latency, total CPU work, replay +or FAST inclusion. Disk work moves to the worker; it does not disappear. The +baseline is the exact previously deployed combined validator, SHA256 +`a58366704680154628ff0a4c6b4027a3e5b79f1909eb39bffeec326c008f0b68`, not the full +branch versus alpenglow-dev. + +## Limits + +The blocked-writer regression establishes isolation of completion publication. +It does not eliminate all spool I/O: ordered invalidations, slot-file operations +and shutdown can still wait for storage. Historical live trials did not establish +an overall large-block p99 improvement and included remaining verification tails. +Keep those limits separate from the component benchmark above. + +[Historical measurements and source](spool-completion-journal-evidence.md) retain +the original deployment comparison and its trace/probe qualifications. diff --git a/docs/shred_retention_performance.md b/docs/shred_retention_performance.md new file mode 100644 index 000000000..ebcb9af82 --- /dev/null +++ b/docs/shred_retention_performance.md @@ -0,0 +1,44 @@ +# Shred retention sweep scheduling + +`SlotAssembler` previously scanned its incomplete/completed slots, block-ID hints, +rejected IDs, partial observations, and priority repair state on every incoming +shred, including duplicates and completed-slot packets. The new schedule skips +age sweeps when neither the observed edge nor the retention floor nor relevant +retained state has changed. The original sweep algorithm and age limits remain. + +Mutations invalidate the cached sweep after adding old identity hints or partial +observations, resetting a generation, or terminating completion. Completion +success, error, cancellation and abort all release protected parent-ID state. +Repair-floor advancement, reduction and clearing are checked on the next packet. +The hard incomplete-slot capacity check remains unconditional on every packet, +including catch-up insertion at an unchanged edge. Eviction preference and +protection for completing generations and the repair head are unchanged. + +Regression coverage includes changing the repair floor without advancing the +edge, every terminal completion outcome, newly added old identity hints, and +capacity overflow at a fixed edge. Existing completion, cancellation, generation, +FEC and repair tests also run as part of the full Turbine suite. + +## Isolated benchmark + +`BenchmarkRetentionRepeatedCompletedShred` drives the public `AddShred` path +with a completed-slot packet and 513 retained entries in each of four metadata +maps. It measures the repeated scan/rejection case, not full packet decoding, +authentication, FEC, replay, or a whole-validator speedup. Five 500 ms runs: + +| Host | Baseline median | Candidate median | Allocation | +| --- | ---: | ---: | ---: | +| Apple M4 Pro | 11,617 ns/packet | 7.718 ns/packet | 0 on both | +| Ryzen 7 9700X, GOMAXPROCS=2 | 11,443 ns/packet | 12.99 ns/packet | 0 on both | + +Baseline and candidate use the same fixture; Go's source overlay selects the old +assembler for baseline runs without changing other source. This fixture shows the +avoided work, not a prediction for a validator's actual map population. + +Validation passed locally and natively: race suites for Turbine, replay, +consensus and node; Turbine vet; complete validator build. The live integration +applies this patch over the exact deployed FEC/peer-isolation/status-expiry source. + +Historical live trials and their limitations are in the +[archived evidence](streaming-preparation-evidence.md). Component results do not establish +a sustained FAST-inclusion improvement. diff --git a/docs/spool-completion-journal-evidence.md b/docs/spool-completion-journal-evidence.md new file mode 100644 index 000000000..d2b40d986 --- /dev/null +++ b/docs/spool-completion-journal-evidence.md @@ -0,0 +1,14 @@ +# Spool Completion Journal: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/f0b72ab239efbbd4811498d73b36ba73eb6192e1) +(tag `review-evidence-20260916-spool-completion-journal`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/f0b72ab239efbbd4811498d73b36ba73eb6192e1/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/status-checkpoint-capture.md b/docs/status-checkpoint-capture.md new file mode 100644 index 000000000..8ba391cb5 --- /dev/null +++ b/docs/status-checkpoint-capture.md @@ -0,0 +1,98 @@ +# Transaction-status checkpoint capture and encoding + +Replay captures immutable lineage and coverage metadata before submitting a +checkpoint to the promotion worker. Sorting, encoding and writing happen on +that worker. Capture does not retain parent links outside the selected window. +Publication and durable-root ordering are unchanged. + +Each node memoizes its canonical encoded body on first serialization. Capture +and pruning share the same cache object when copying a node header; they never +copy a used synchronization primitive. Encoding depends on the immutable slot, +block-ID presence/value and status delta, not its parent link. Concurrent +encoders synchronize through `sync.Once` without taking the live cache lock. +Each snapshot still constructs its own coverage header and returns an owned +output buffer. The MTS2 format and restore validation are unchanged. + +The cache retains roughly one extra encoded window (30 MB for 1.5 million +keys), plus any nodes pinned by older views. There is no global encoding map: +caches become collectible with their last node/view. A completely new window +still pays for all sorting. Output copying and checkpoint I/O remain necessary. + +## Encoding benchmark + +`BenchmarkTransactionStatusCheckpointEncoding` uses a 300-root window with +5,000 keys per root (1.5 million keys, roughly 30 MB encoded). Each iteration +replaces the specified number of roots. Fixture creation and initial warming +are excluded; new node headers, sorting and output allocations are included. +The baseline is the original uncached wire encoder retained in tests. + +Apple M4 Pro, Go 1.26.4, one caller, GOMAXPROCS=12; medians of three runs: + +| New roots per checkpoint | Original encoding | Cached encoding | +| --- | ---: | ---: | +| 1 | 158.05 ms | 1.35 ms | +| 8 | 157.14 ms | 5.09 ms | +| 32 | 155.62 ms | 17.51 ms | +| 128 (default fold cadence) | 153.74 ms | 67.11 ms | +| 300 (entirely new) | 155.90 ms | 156.29 ms | + +At the default cadence, allocated bytes per encoding fell from 99.12 MB to +57.33 MB; this excludes retained heap. These are encoding measurements, not +end-to-end fold/replay timings or live FAST improvements. Data distribution +matters: newly rooted large blocks can account for most keys in the window. + +Run `go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusCheckpointEncoding$' -benchmem -benchtime=1s -count=3`. + +Tests compare exact bytes with the original encoder across coverage flags, +block IDs and sorted groups; check concurrent encoding during pruning/unwind; +verify cache sharing before and after warming; and restore checkpoints after +callers mutate their own output buffers. The replay race suite and vet pass. + +Related behavior: [status expiry](transaction-status-expiry.md) and +[status publication](transaction-status-publication.md). + +## Native Zen 5 validation + +AMD Ryzen 7 9700X, Go 1.26.4, GOMAXPROCS=2, Nice 15 and a two-core CPU quota, +while the validator continued its normal workload. Same moving-window fixture; +three samples per case, medians below. This compares the original uncached +encoder with memoization, not the whole status-publication change against dev. + +| New roots per checkpoint | Original encoding | Cached encoding | +| --- | ---: | ---: | +| 1 | 195.34 ms | 3.73 ms | +| 8 | 195.15 ms | 8.56 ms | +| 32 | 194.77 ms | 23.54 ms | +| 128 (default fold cadence) | 194.52 ms | 85.02 ms | +| 300 (entirely new) | 201.88 ms | 198.55 ms | + +The default-cadence result is approximately 2.3x, with the same 99.12 → 57.33 MB +allocation reduction. Cold/all-new windows remain roughly unchanged. Native +combined race suites, vet and the validator build passed. These are historical staging measurements. They do not establish an isolated +live reduction in durable-root lag or missed FAST votes. + +## Fold admission before collecting account writes + +Replay checks for a checkpoint batch on every iteration, including skipped +slots. `WorkingSet.PromotionChunk` first counts eligible held slots under its +read lock. If fewer than the configured batch size are available, ordinary +admission returns nil without allocating account-pointer lists. When ready, +it collects only the oldest batch, not the entire eligible suffix. Forced +partial folds still collect the available prefix. + +This preflight is not a finality shortcut or a new recovery policy. Replay's +existing finality/verification gates supply the upper bound. Selection and +collection hold the same lock; account pointers retain their existing ownership +contract. Preparation does not prune the suffix or advance the durable root. +The worker's write/commit order, required resume context, checkpoint reference +validation, completion bookkeeping, and forced shutdown/epoch-boundary paths +are unchanged. + +`BenchmarkBuildFoldJobWaitingForBatch` holds 127 slots with 512 account writes +each while waiting for the default 128-slot batch. On Ryzen 9700X, +GOMAXPROCS=8, three 300 ms runs, median admission-check time fell from 426 µs +to 31.8 ns; 627,008 bytes and 134 allocations per rejected preparation became +zero. This measures an ineligible batch check, not encoding, disk I/O, or a +ready checkpoint. Boundary tests cover gaps, the finality upper bound, a full +batch, forced partial batches, and selection after promotion; existing replay +checkpoint/recovery tests cover the unchanged durable path. diff --git a/docs/status-checkpoint-expiry-evidence.md b/docs/status-checkpoint-expiry-evidence.md new file mode 100644 index 000000000..978180120 --- /dev/null +++ b/docs/status-checkpoint-expiry-evidence.md @@ -0,0 +1,19 @@ +# Status Checkpoint Expiry: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf) +(tag `review-evidence-20260916-status-checkpoint-expiry`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. + +The tagged snapshot also preserves the later 64-partition visible-status-map +experiment. That experiment is deliberately excluded from this review: it added +preparation work and did not demonstrate an overall large-block p99 benefit. +Earlier publication preparation and immutable-node encoding reuse remain. diff --git a/docs/streaming-preparation-evidence.md b/docs/streaming-preparation-evidence.md new file mode 100644 index 000000000..53e444472 --- /dev/null +++ b/docs/streaming-preparation-evidence.md @@ -0,0 +1,14 @@ +# Streaming Preparation: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/1c1171d3661d0404b013a9bf9391e23eb660706e) +(tag `review-evidence-20260916-streaming-preparation`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/streaming_message_identities.md b/docs/streaming_message_identities.md new file mode 100644 index 000000000..d852b22d4 --- /dev/null +++ b/docs/streaming_message_identities.md @@ -0,0 +1,106 @@ +# Prepare transaction message identities during shred arrival + +The live Zen 5 admission trace found ~12.45 ms of message serialization, +hashing and identity-cache preparation after assembly on large blocks, followed +by ~1.66 ms of same-block duplicate-plan work. Most blocks reached this stage +while replay was already waiting. + +Turbine now asks signature-verification workers to retain a message identity +derived from the exact canonical bytes they already serialize for verification. +The existing two workers process available groups without waiting for more +transactions. TPU callers of the ordinary verifier do not compute these extra +identities. + +Each opaque result binds a successful signature verdict to the transaction +pointer, message version and recent blockhash. Completion joins requests and +imports only identities covering the final ordered transaction slice. Canceled +requests are reverified; discarded UpdateParent prefixes are excluded. Results +are caller-owned and cannot refer to reusable verifier scratch. The block cache +owns its imported storage and remains nonserialized. The existing requirement +that signed message contents remain immutable still applies; arbitrary in-place +message edits require invalidation, as before. + +Same-block duplicate rejection and mutable ancestor/status checks remain in +place. This change does not cache the duplicate-check plan or alter epoch +processing, vote persistence, worker counts, scheduling deadlines, or packing. + +## Zen 5 comparison + +Baseline: the currently deployed shred-retention source, including its existing +FEC and peer-isolation integrations. Candidate: that same source plus streaming +identities. Both binaries use the identical new benchmark harness. + +`BenchmarkEntryMessageIdentityArrival` feeds real generated data shreds through +the assembler, entry prefetch and signature-verification pipeline, then includes +the admission-time identity lookup. Each block has 33,760 single-signature +transactions, either 228 or 1,232 bytes on the wire. The tip model schedules +component arrivals across 200 ms; catchup offers every shred immediately. +Fixture construction, packet parsing and shred-signature authentication are +outside the timer. There is no network loss, transaction execution, PoH/reward +processing or final whole-block duplicate map in this benchmark. + +AMD Ryzen 7 9700X, Narya `r51` AVX-512 backend, two signature workers, target eight +signature lanes, GOMAXPROCS=16. Five iterations per scenario per run, two runs +per variant in baseline/candidate/candidate/baseline order. Values below are the +median of the two run medians, not a percentile computed over all ten samples. +The live validator and load services continued running, so host contention can +affect the results. + +| Arrival model | Transaction size | Last shred → identities available, baseline | Candidate | +| --- | ---: | ---: | ---: | +| Over 200 ms | 228 bytes | 13.62 ms | 2.66 ms | +| Over 200 ms | 1,232 bytes | 73.72 ms | 7.81 ms | +| Catchup | 228 bytes | 92.90 ms | 88.40 ms | +| Catchup | 1,232 bytes | 154.75 ms | 118.20 ms | + +The table includes completion work; it does not merely move that work out of +the admission timer. Admission's identity lookup alone falls from ~12.1 to +0.24 ms for 228-byte tip transactions and ~66.9 to 0.26 ms for maximum-size tip +transactions. The block-wide duplicate check remains additional work. + +CPU time per tip block was 189.75→197.20 ms for the small fixture (~4% higher) +and 316.40→304.80 ms for the maximum-size fixture (~4% lower). The intended gain +is less work after arrival, not a blanket claim of lower CPU. Maximum-size +fixtures allocate roughly 37 MB fewer bytes per block by avoiding a second +message serialization; retaining opaque identities also has a memory cost. + +The initial runs without an explicit backend used the library's unconfigured +default and are excluded from these deployment-relevant results. + +## Validation and current limits + +Local affected-package tests, race tests for txverify/block/turbine/replay, Go +vet, and a full validator build passed. Tests cover legacy/v0/v1 canonical +identities, failed signatures, scratch reuse, pointer/order/blockhash mismatch, +storage ownership, JSON round-trips, early preparation and mixed canceled-batch +fallback. Existing turbine tests cover invalid and discarded prefixes and +completion cancellation. + +Historical deployment observations are retained in the +[archived evidence](streaming-preparation-evidence.md). Their unequal workloads +and epoch boundary do not establish an isolated FAST-score improvement. + +## Recovery from inconsistent prefetch metadata + +Missing/oversized retained ranges, partial identities and identity-binding +mismatches now log a warning and fall back to full signature verification of the +final block. Bounds are checked before slicing the final transaction array. Old +readers are joined first, including on cancellation. Valid final transactions +can therefore recover from an optimization bookkeeping fault; a successful +cached verdict for different bytes cannot authorize the final transaction. +Normal cached signature failures remain errors, as do failed re-verification, +cancellation and a closed verifier. No signatures or duplicate checks are skipped. + +Regression tests cover valid and invalid final blocks for each metadata fault, +cache identities matching the final transaction order, canceled-reader ownership +and verifier failure. The ordinary path retains exact-range/byte checks and +verified identity reuse. Full re-verification is exceptional and costs additional +work; this change is a correctness/availability fix, not a throughput claim. + +The `[sigverify]` starter configuration and its test moved here from the voting +branch, because this branch reads those keys and supports the legacy backend key. +The unrelated explicit `tuning.use_pool` template setting is omitted: enabling +pooling by default and fixing retained vote ownership belong to the runtime +branch. In the combined build, its configuration defaults still enable pooling. +The default remains two Turbine verification workers (bounded by GOMAXPROCS). +Many-core catch-up tuning remains a separate measurement question. diff --git a/docs/transaction-status-expiry.md b/docs/transaction-status-expiry.md new file mode 100644 index 000000000..538acb0f8 --- /dev/null +++ b/docs/transaction-status-expiry.md @@ -0,0 +1,61 @@ +# Batched transaction-status expiry + +Applying an asynchronous checkpoint still calls `TransactionStatusCache.Root` +on replay. Live Zen 5 instruction probes measured 43–101 ms inside that function. +The previous expiry path visited every key in every retired bank, even when an +entire recent-blockhash group could be discarded. + +Expiry now examines the expired and retained bank deltas by blockhash. It drops +fully expired groups directly. For a group spanning the cutoff, it either +subtracts the expired keys or rebuilds the visible reference counts from the +retained deltas, whichever requires fewer key visits. Retained unrooted banks +are included. Physical map reclamation is still Go GC work; this is not a claim +that memory reclamation costs disappear. + +The 300-root retention rule, immediate logical expiry, duplicate-key reference +counts, selected-parent validation, checkpoint format and immutable producer +views are unchanged. All index changes remain under the existing cache lock. +This does not move unsafe mutable state to another goroutine or delay expiry. +A long-lived blockhash with many transactions on both sides of the cutoff can +still require substantial per-key work. This patch reduces that work to the +smaller side; it does not give a constant-time worst-case bound. + +## Validation + +The replay race suite, replay vet and validator production build pass. New tests +compare exact visible indexes against the original per-key removal for 100 +random lineages with shared hashes, collisions and empty groups, then unwind +surviving banks. A Root integration test checks pinned producer views, +checkpoint bytes, restored duplicate detection and rooted-unwind rejection. + +M4 Pro, Go benchmark, single caller, two iterations per case. Each iteration +expires 128 banks of 33,760 unique keys (4,321,280 entries) and retains another +33,760 entries. Setup is outside the timer. The baseline invokes the original +per-key removal; the new path invokes batched expiry. These are **expiry-path** +measurements, not end-to-end Root/replay or a prediction of live FAST scores. + +| Recent-blockhash grouping | Old expiry | Batched expiry | +|---|---:|---:| +| Groups shared by four expired banks | 185–189 ms | 0.037–0.080 ms | +| One fully expired group | 604–614 ms | 0.025–0.026 ms | +| One group shared by expired and retained banks | 590 ms | 1.63–2.36 ms | + +Run `go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusBatchExpiry$' -benchtime=1x -count=2`. + +## Native benchmark + +Ryzen 7 9700X, Go 1.26.4, original per-key expiry versus batched expiry. +Benchmarks ran with GOMAXPROCS=2, nice=15, one caller and three iterations per +case, while the validator and loader remained active. Setup and later GC are +excluded from the expiry timer. Each case expires 4,321,280 entries (128 banks +of 33,760) and retains 33,760 entries. These synthetic batches exceed the earlier +live stall samples and are not an end-to-end replay or FAST-score comparison. + +| Shape | Original expiry | New expiry | +|---|---:|---:| +| Four-bank blockhash groups | 306–311 ms | 0.049–0.057 ms | +| One fully expired blockhash group | 700–718 ms | 0.024–0.031 ms | +| Group crossing the retention boundary | 717–735 ms | 1.85–2.05 ms | + +[Historical evidence](status-checkpoint-expiry-evidence.md) preserves the original +source revisions, raw measurements and validation. diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md new file mode 100644 index 000000000..68ca309fa --- /dev/null +++ b/docs/transaction-status-publication.md @@ -0,0 +1,73 @@ +# Preparing transaction-status publication during execution + +Replay previously built the immutable per-bank transaction-status delta and grew the visible duplicate index only after execution and bank-state publication. In a prior live sample of 25 large blocks, TransactionStatusCommit took 7.704 ms median and 10.206 ms maximum. Those live timings motivate this change; they are not the controlled benchmark baseline below. + +Count identities by recent blockhash and allocate each delta map at its final capacity. Pre-size newly created visible maps too. For banks with more than 32 transactions and GOMAXPROCS greater than one, prepare the immutable delta during account loading and execution. Smaller banks and single-thread configurations keep the work inline. There is at most one preparation task per ProcessBlock call, and every return joins it, including rejected banks. No status becomes visible during preparation. + +The worker reads immutable prepared message identities and briefly snapshots only blockhash slice offsets under the cache read lock. It builds its private maps outside the lock. Commit checks exact block/identity binding, complete coverage and parent lineage under the publication lock. It rechecks ancestor duplicates unless the successful pre-execution validation belongs to the same cache instance, immutable identity set and unchanged cache version (see below). A changed slice offset, disappearance of a previously nonzero-offset group, or mismatched preparation triggers a rebuild from the actual block's identities. Publication still happens only after successful bank-state commit. Failed instructions within an accepted bank remain processed; rejected banks publish nothing. Pinned views, snapshots, reference counts and unwind keep their existing semantics. + +TransactionStatusPreparation measures worker wall time, which overlaps execution; it is not additive with replay wall time. TransactionStatusPreparationWait measures the residual join and is nested inside TransactionStatusCommit. The latter still includes waiting, final checks, visible-index updates and node publication. Preparation time excludes initial goroutine scheduling delay; any residual scheduling delay remains in the join/commit timer. + +## Native benchmark + +AMD Ryzen 7 9700X (Zen 5), Go 1.26.4, GOMAXPROCS=2. Tests ran in a separate process on the validator host with Nice=15 and a 200% CPU quota; the validator and loader continued running. This is a shared-host microbenchmark, with observable timing variation. Five samples per case, ten iterations per sample; values below are medians of sample means, not per-block percentiles. + +Each block has 33,760 unique prepared message identities spread across one or four recent blockhashes. Existing-group cases seed 33,760 different ancestor transactions. Fixture creation, hashing, seeding and unwind are untimed. Existing maps retain capacity after unwind: the first timed commit's growth is amortized across the ten iterations. This does not model an index growing indefinitely across live blocks. + +The frozen baseline functions exactly match alpenglow-dev commit `33dde4050d9250557583395810799aaac2f54017`. Both versions use the same prepared identities, parent/duplicate checks and fixtures. + +| Recent blockhash groups | Parent has keys in these groups | Baseline commit | Sized maps, inline | Preparation + commit, no overlap | Commit after preparation | +|---|---|---:|---:|---:|---:| +| 1 | No | 4.990 ms | 3.332 ms | 3.393 ms | 1.587 ms | +| 1 | Yes | 4.991 ms | 4.657 ms | 4.434 ms | 2.757 ms | +| 4 | No | 4.625 ms | 3.744 ms | 5.916 ms | 2.205 ms | +| 4 | Yes | 5.224 ms | 4.644 ms | 4.949 ms | 2.858 ms | + +The last column deliberately excludes delta preparation: it measures the work remaining if execution hides preparation completely. It is not total replay or CPU work. Total publication allocations with new groups fell from approximately 6.30 MB to 3.15 MB per block. Existing-group allocation figures include the amortized first growth described above. + +The four-new-group total-work sample was slower. Preserve that result rather than claiming improvement in every sample. A subsequent baseline/candidate/candidate/baseline comparison of that same case, with 50 iterations per sample, measured baseline **4.400 and 4.565 ms**, candidate **2.985 and 3.131 ms**. This supports a reduction in work but does not isolate the cause of the earlier timing variation. + +## Execution contention and small blocks + +A separate controlled benchmark performs 4,096 load-and-execute calls using the existing transfer fixture while preparing 33,760 independent status keys. It does not commit transfer accounts, and its status fixture differs from the repeated transfer fixture. It tests scheduling/allocation contention, not whole-block replay or a valid block workload. + +With two Go execution threads, the final implementation measured **20.678 ms baseline**, **18.574 ms with sizing alone**, and **17.091 ms with overlap**. Execution itself measured 15.070, 14.369 and 14.967 ms respectively. Thus preparation competed with execution relative to sizing alone, but the shorter final stage outweighed that cost in this controlled workload. These are separate medians and need not add exactly. + +The initial unrestricted version showed no additional total-time benefit from overlap with GOMAXPROCS=1. Tiny-block measurements also showed roughly a microsecond of avoidable scheduling overhead. The final implementation therefore does no background preparation with one Go execution thread or at most 32 transactions. Empty and one-transaction cases retain the baseline allocation counts. The 32-transaction case benefits from sizing without launching a worker. Threshold and single-thread behavior have regression coverage. + +## Validation and limits + +Full replay and block race suites passed on both Zen 5 and M4 Pro. Metrics has no tests. Native vet for replay/metrics and the validator build passed. Tests cover fork replacement introducing a duplicate after preparation, concurrent sibling publication, stale identity binding, changed snapshot slice offsets, rejected/incomplete banks, mismatched preparation, pinned views, snapshot restore, unwind, empty banks and scheduling boundaries. + +Raw logs, source hashes, summaries and the alternating recheck are in [results/status-publication/2026-09-15](https://github.com/Overclock-Validator/mithril/blob/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf/docs/results/status-publication/2026-09-15). The baseline comparison covers only status publication. No live replay or FAST improvement is claimed. The staging binary was not deployed; the existing validator remained active and voting throughout the tests. + +Reproduce from this branch: + +```sh +GOMAXPROCS=2 go test -race -p 2 ./pkg/replay ./pkg/block ./pkg/metrics -count=1 +GOMAXPROCS=2 go vet -p 2 ./pkg/replay ./pkg/metrics +GOMAXPROCS=2 go build -p 2 ./cmd/mithril +GOMAXPROCS=2 go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusPublication$' -benchtime=10x -count=5 +go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatus(ExecutionOverlap|SmallPublication)$' -benchtime=100ms -count=5 -cpu=1,2 +``` + +## Reusing pre-execution ancestor validation + +`ProcessBlock` now carries a private validation receipt from its successful ancestor scan to status publication. Under the commit lock, an unchanged receipt avoids scanning all transaction messages again. Publication still checks block binding, complete coverage and parent lineage every time; a missing, foreign or stale receipt performs the full ancestor scan. Direct `CommitBlock` callers retain the full scan. + +The receipt is bound to the cache instance and exact immutable prepared-identity pointer. Visible-index insertion/removal, tip binding, root/prune and restore invalidate the version, including empty commits. Committing and then unwinding back to an identical parent cannot revive a receipt. Version saturation disables reuse permanently rather than wrapping. Snapshot/Agave recovery creates a new cache instance. Receipts are never persisted, and no checkpoint format, durability, voting-resume or crash-recovery guarantee changes. + +The publication benchmark adds `validated_commit` and `invalidated_commit` alongside `prepared_commit`. All three exclude delta preparation and the pre-execution scan. The first reuses that scan; the second calls `Root` between validation and publication, forcing revalidation. Each iteration unwinds and obtains a fresh receipt outside the timer. These are incremental publication comparisons, not the full PR against alpenglow-dev or per-block tail latency. Tests exercise fork replacement introducing duplicates, concurrent sibling commits, cross-cache and cross-identity misuse, snapshot replacement, pruning/root invalidation, binding changes, transaction replacement and version saturation. + +Zen 5 incremental measurement (Ryzen 9700X, Go 1.26.4, GOMAXPROCS=2, five samples × 20 iterations, Nice=19 / 200% CPU quota on the running validator host): + +| Recent blockhash groups | Existing ancestor groups | Full recheck | Reused validation | Invalidated validation | +|---|---|---|---|---| +| 1 | yes | 2.510 ms | 1.364 ms | 2.536 ms | +| 4 | yes | 2.412 ms | 1.283 ms | 2.440 ms | +| 1 | no | 1.461 ms | 1.472 ms | 1.769 ms | +| 4 | no | 1.323 ms | 1.395 ms | 1.303 ms | + +Values are medians of sample means. Existing-group cases remove approximately 1.1 ms of repeated lookup work; new-group cases show no clear gain and shared-host variation. Full native replay/block race suites, targeted node recovery race tests, vet and the combined build passed. Local replay race tests and vet also passed. + +Original run artifacts are retained in the [evidence archive](status-checkpoint-expiry-evidence.md). diff --git a/docs/transaction_sigverify_streaming.md b/docs/transaction_sigverify_streaming.md new file mode 100644 index 000000000..4c40da2bb --- /dev/null +++ b/docs/transaction_sigverify_streaming.md @@ -0,0 +1,302 @@ +# Transaction signature verification during shred arrival + +Turbine verifies transaction signatures as complete entry batches arrive. The +default shared transaction pool has `min(2, GOMAXPROCS)` workers and targets eight signature +lanes. A ready batch containing four transactions runs immediately; there is no +timer or minimum occupancy requirement. The same policy handles live reception +and repair catch-up without a mode transition or a 200 ms batching delay. + +## Order of work + +1. The receiver authenticates the shred and performs existing assembly/FEC + recovery. The packet reader advances a contiguous data-shred frontier and + attempts a nonblocking background enqueue. Assembly retains an authenticated + root and a snapshot of its source shred for each FEC set when available. +2. Two background preparation workers copy and decode complete DATA_COMPLETE + entry batches. A batch may span several FEC sets. They submit its immutable + transactions to the shared signature pool while later shreds arrive. +3. Signature groups contain available transactions up to the configured lane + target. Transactions with multiple signatures remain indivisible. Large + already-decoded requests bundle four vector groups per dispatch job (normally + 32 one-signature transactions). Requests smaller than + `2 * workers * batch_target * 4` transactions keep one group per job; the + default threshold is 128 ready transactions. A rolling per-request window + refills when any job finishes, without a wave barrier or batching timer. +4. Once the full slot is assembled, completion compares shred slices directly + against the cached component bytes, including padding. Cache hits avoid a + second component buffer; misses allocate a fresh buffer at its exact size. + Complete contiguous slots supply shred order directly, avoiding two sorts. + Completion processes all Alpenglow markers and FEC roots and constructs the + final ordered block. It submits any transactions not already covered and joins + signature work for the retained batches before marking the block verified. +5. The existing replay pipeline receives the verified block. This change does + not execute transactions before full-slot admission, alter execution batching, + or move parent-dependent validation ahead of its required state. + +The 200 ms slot interval provides an opportunity to overlap work. It is not a +mandatory local wait: replay can continue as soon as the block is available and +its checks complete. If all shreds arrive in a burst during catch-up, full-slot +completion uses the same efficient groups with whatever early work was able to +start. Four workers can improve catch-up latency but occupy more cores at once. + +## Bounds and correctness + +- At most eight slot generations hold early preparation reservations, with a + combined 64 MiB budget for raw component bytes and a 1 MiB per-component limit. + Decoded transaction objects add heap overhead. Saturation skips optional early + work; normal full-slot verification still covers every retained transaction. +- One queued/active preparation token per generation coalesces packet arrivals. + Verifier requests and each request's outstanding jobs are also bounded. + Request admission can wait behind existing requests; this is not a strict + replay-head priority scheduler. +- No transaction decoding or verifier admission occurs under the assembler mutex or on the + packet reader. A gap prevents early component decoding until recovery or + arrival closes it. Duplicate shreds do not create duplicate requests. +- Cached results belong to one generation, shred range and exact byte sequence. + Reset/eviction cancels that generation. Reservations remain charged until + admitted readers have relinquished their transaction buffers. +- A FEC root cache retains at most one source snapshot per FEC state, in addition + to the entry-prefetch budget. Completion preserves the deterministic choice of + the lowest-index non-recovered data proof, then lowest coding position. Cached + roots require the same source, parsed root inputs, and exact payload bytes; + mismatches, unauthenticated callers, and spool hydration recompute the root. +- `UpdateParent` can discard an optimistic prefix. Parse and marker checks still + cover that prefix, while its transaction signature verdict is discarded along + with its transactions. Retained signatures must all pass before replay. +- Cancellation is not an invalid-signature verdict. A retry on the same slot + generation verifies transactions again if an earlier request was canceled. + An admitted job finishes its first vector group; cancellation can skip later + groups in that job. The request joins all admitted jobs before releasing input. + +## Configuration and observability + +```toml +[sigverify] +backend = "auto" +workers = 0 # min(2, GOMAXPROCS) +batch_target = 8 # 4 or 8; short groups never wait to fill +disable_shred_overlap = false +``` + +Equivalent CLI flags are `--sigverify-workers`, `--sigverify-batch-target` and +`--sigverify-disable-shred-overlap`. These settings apply to Turbine transaction +signatures; TPU, shred signatures, consensus BLS and replay's fallback verifier +retain their existing configuration. + +Use `TurbineFullToReady` to measure residual wall time after the slot becomes +complete. `TurbineEarlyVerifiedTransactions` counts retained transactions whose +verification finished before full assembly. `TurbineTransactionSigverify` now +measures completion's outstanding-signature join/fallback work. The remaining +wait for an already claimed background preparation job, including +any outstanding admission delay, appears in `TurbineEarlyPreparationWait`, +separately from active completion decoding. It does not sum every background +admission wait. Early parse and signature durations are component sums observed +at completion; +a discarded prefix still being verified is not included. These durations +overlap reception and each other and must not be added as sequential stages +or interpreted as CPU time. + +## Benchmark scope + +`BenchmarkTransactionVerificationFlow` compares two/four workers and targets +four/eight using catch-up, synthetic 200 ms component arrivals, and sparse +four/seven/eight-transaction components. It supports captured public transaction +fixtures and generated, distinct valid transactions of exactly 228 or 1,232 wire +bytes. Decoding and fixture construction are outside those pool measurements. + +The execution contention probe runs the real transfer load/execute benchmark +alongside signature work on the same eight physical cores. Its separate +processes measure hardware/OS contention, excluding shared Go scheduler/heap +effects, full-block dependency planning, commit and live network timing. Pool +throughput alone is insufficient evidence of end-to-end replay improvement. + +Measured Zen 5 results, raw logs, and validation details are in the +[September 12 benchmark report](https://github.com/Overclock-Validator/mithril/blob/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results/sigverify-streaming/2026-09-12-zen5/README.md). +The subsequent [direct cache-comparison report](https://github.com/Overclock-Validator/mithril/blob/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results/sigverify-direct-cache/2026-09-12-zen5/README.md) +isolates the removal of redundant component-buffer construction at completion. +The [completion follow-up report](https://github.com/Overclock-Validator/mithril/blob/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results/completion-followup/2026-09-12-zen5/README.md) +measures direct ordering, authenticated-root reuse, and the four-vector job policy. + +The [standalone PR review](https://github.com/Overclock-Validator/mithril/blob/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results/streaming-pr-review/2026-09-13/README.md) +records extraction onto current `alpenglow-dev`, the small shared component-boundary +prerequisite, final allocation improvement, and the scope of the live trial. + +## Worker default compatibility + +The automatic transaction-verifier default changes from `(GOMAXPROCS + 1) / 2` workers to +`min(2, GOMAXPROCS)`, including when shred overlap is disabled. This favors spare +CPU capacity for execution and other verification at the tip. It is not a claim +of maximum catch-up throughput on every core count or backend. Set +`--sigverify-workers N` or `[sigverify] workers = N` explicitly when tuning a +larger machine; disabling overlap alone does not restore the previous worker +count. Existing two/four-worker contention measurements are in the September 12 +report above. No 32-core comparison was performed. + +## Reserved admission for completion + +Decoded prefetch components now use a separate admission class. The existing total request limit remains `2 * workers`; at most `2 * workers - 1` requests may prefetch. Thus the default two-worker pool keeps four total permits, with at most three occupied by prefetch. Waiting completion/full-block recovery requests win the next free permit over prefetch. Their admission, cancellation and close registration share the verifier mutex; notification channels are allocated only when callers must wait. + +No worker or job queue is added, and verification, vector width, job grouping and per-request rolling windows are unchanged. Accepted jobs finish normally and are joined before transaction memory can be reused. No signature checks are skipped. Prefetch may wait while completion callers remain queued and resumes when that backlog drains. This is completion-class priority, not exact replay-head priority: future-slot completions also qualify, and already admitted prefetch work is not promoted or preempted. Checkpoint, persistence and voting-recovery contracts are unchanged. + +### Saturation benchmark + +`BenchmarkVerifierCompletionReservation` verifies four already-ready prefetch components (256 or 4,096 signed 228-byte transactions each) plus a 32-transaction completion request. All requests are joined; total-work timing includes all four components. Two verifier workers, eight signature lanes, four vector groups, GOMAXPROCS=8, Narya r51, Ryzen 9700X / Go1.26.4. Three runs of 100 iterations each, Nice19 and a 200% CPU quota on the shared validator host. Test intervals are excluded from live FAST comparisons. + +The before comparison uses the previously deployed source (status-validation combined build, SHA256 `2c81fc403e8a6eca73b87ede51890041347e1af0e3c63df005fcfeb26448e872`) with only the benchmark added via a Go test overlay. The candidate also measures the shared-class control to distinguish policy from incidental overhead. These are incremental admission results, not the whole PR versus alpenglow-dev. + +For four 4,096-transaction components, medians of the three per-run statistics were: + +| Measurement | Previous deployment | Reserved admission | +|---|---:|---:| +| Completion admission p50 | 37.19 ms | 0.000742 ms | +| Completion admission p99 | 45.78 ms | 0.004599 ms | +| Completion finished p99 | 47.31 ms | 1.488 ms | +| All work finished p50 | 39.43 ms | 39.11 ms | +| All work finished p99 | 49.09 ms | 50.65 ms | + +Completion-finished p99 ranged 46.43–54.52 ms before and 1.477–1.719 ms after. Admission p99 ranged 44.27–53.76 ms before and 0.003206–0.06401 ms after. Every iteration reached its intended request occupancy. For 256-transaction components, completion-finished p99 medians were 3.215→1.165 ms. Shared-host scheduling introduces variation; reserving admission does not remove queued-job or CPU delays, and these 100-sample tails are not a live p99/FAST claim. Total-work throughput was roughly unchanged; no total-work tail improvement is claimed. + +Original run artifacts are retained in the [evidence archive](streaming-preparation-evidence.md). + +### Completion-critical shred diagnostics + +The optional `MITHRIL_ENTRY_TRACE_MOD`, `MITHRIL_ENTRY_TRACE_SECONDS` (at most +1,800), and `MITHRIL_ENTRY_TRACE_FILE` settings are read at process startup. +Use a new output file for each capture. Tracing is disabled by default. + +Large-block batch records include `critical_shred_index` and its admission +source: `non_repair`, `repair`, or `fec_recovery`. The critical index maximizes +local availability time across the batch **and its preceding DATA_COMPLETE +boundary**. Equal timestamps select the lowest index and report the tie count; +unknown coverage still sets `availability_known=false`. This identifies the +last locally available dependency, not necessarily the replay cursor's current +blocking range. Correlate with execution groups before calling it a replay stall. + +Recovered shreds identify the triggering packet's index, FEC set, coding/data +type and repair status. `non_repair` can include spool hydration. Admission entry +timestamps precede the assembler lock; they are not socket/NIC timestamps. A zero +admission-entry timestamp means it was not sampled. Admission-to-availability +includes local processing and, for recovered data, reconstruction. + +The same JSONL file also contains `event="repair_send"` records. Consumers must +separate these from block reports. Join by `origin_unix_ns`, slot and shred index, +then order by send timestamps; attempt IDs can reset. Start/end bracket the UDP +write, and `success` means only that the local write succeeded. Highest-index +probes are explicitly marked and must not be treated as exact-index requests. +Request records can exist for slots without a large-block report. Send records include the peer endpoint and nonce for response correlation. + +Both queues are bounded and producers never wait for the writer. The cumulative +`dropped_reports` counter covers queue drops and encoding failures; an absent +repair record is not proof of no request if records were dropped. Tracing does +not change repair scheduling, retry intervals, fanout or verification checks. + +### Repair ordering for streaming + +When a streaming subscriber is installed, the first priority repair slot puts +its earliest missing data span ahead of the usual cheapest-FEC-unlock ordering. +Known FEC sets still request only their recovery deficit; an unknown-layout hole +prioritizes one missing index without guessing its FEC shape. Remaining work, +other priority slots and freshness repair retain their previous ordering. +Request budgets, admission shares, retry intervals and fanout are unchanged. + +This trades completing cheap later sets first for making the contiguous input +prefix available sooner. It helps when request capacity is constrained; it does +not accelerate requests already in flight, guarantee an earlier full block, or +prove improved voting latency. The subscriber and priority head are used as the +scope; the selector does not read the execution cursor. + +`go test ./pkg/turbine/repairsim -run TestStreamingPrefixRepairUnderLimitedBudget -v` +compares both policies using authenticated generated shreds and production FEC +recovery, at fixed request budgets and a 20ms simulated round trip. It checks +identical assembled entries/transactions, request counts and full completion, +and measures availability of the first data span. It does not model production +retry timers, peer loss, execution timing, or reproduce a captured live slot. + + +Response-effectiveness tracing also emits `repair_response` and +`repair_admission` events. Treat every record with an `event` field as an event, +not a block report. Match sends/responses by origin, peer, nonce and requested +slot/index; use timestamps to disambiguate nonce reuse. Responses record the +returned index, request-registration timestamp and whether the request had +expired. Registration precedes signing/write; use `send_start_ns` for the closer +approximation to network elapsed time. A matched response is not proof that +assembly accepted it: later receive-path checks can still reject it. + +Admission events cover sampled matched-repair shreds reaching an active assembly; +join to responses by slot/returned index and chronology (no nonce is carried into +the assembler). They report accepted/duplicate/rejected, the coding-layout +recovery deficit before/after, and the number of reconstructed data shreds. +Deficit `-1` means unknown layout; zero means enough shards, not necessarily +successful recovery. Admission timestamps are local assembler entry/exit, +including lock wait and processing, not NIC timestamps. Already-completed, +evicted, or completing slots return before this instrumentation. An unmatched +or canceled response is not emitted as a matched response. Missing records, +particularly with drops or capture boundaries, cannot establish packet loss. + +### Bounded child repair lookahead + +Replay supplies its exact streaming generation as a repair anchor. The +asynchronous decoder can then recognize a decoded header for the immediate next +slot naming that parent, even while replay executes a parent group. A header +published earlier is recovered from the assembler's ready batches. The hint +permits fetching only; it does not establish fork choice, validate the final +parent block ID, or permit child execution before parent completion. + +After ordinary priority and freshness repair, leftover tokens may request up to +four missing data shreds from the child's earliest incomplete FEC span. Existing +in-flight requests for that child count against the four-request lookahead +allowance. Normal repair can independently exceed that allowance. Global rate, +per-scan and admission limits remain in force; lookahead uses bulk single-attempt +policy, with no new retry/fanout or highest-index probing. If no capacity remains, +the child waits. + +Only one child is tracked. The anchor expires after two seconds and is cleared +on parent finalize/discard, parent reset/update, or stream unsubscribe. Child +completion/reset and changed parent markers invalidate its hint. Generation +checks reject stale headers. Already-sent requests still use normal response and +expiry handling. Notifications and repair wakeups remain nonblocking/coalesced; +no extra workers or polling loop are introduced. + +### Highest-index repair followups + +A matched highest-index response triggers followup selection only after receiver +admission and FEC recovery. The assembler supplies its current deficit-aware +selection (at most 256 data requests), rather than treating the interval below +the response as missing. Completed, completing, evicted and absent assembler +slots produce no immediate followups. During disk-only catchup, selection waits +for hydration instead of blindly fetching data that may already be spooled. + +Followups retain the shared token bucket, admission limits, bulk retry policy, +and a reserved token for continued highest-index discovery when needed. A +snapshot can still race with subsequent arrivals; this removes known redundant +requests, not every possible duplicate. No assembler lock is held while signing +or sending requests. Response matching and peer credit are unchanged. + + +### Bounded speculative verification waits + +Replay joins a streaming group's signature verification with one shared 100 ms +budget, capped by the stream's remaining open lifetime. The watchdog stage is +`streaming_sigverify_wait`. Expiry discards the speculative overlay with reason +`sigverify_timeout`; it is not a signature verdict. Whole-block replay still +requires normal verification before accepting the block. + +The streaming wait only observes immutable verifier results. Timing out does not +cancel the shared request or wait for its workers: turbine continues owning its +transactions and retains reservations until readers finish. The owning completion +and cleanup paths retain their joining waits. This bounds speculative replay's +wait, not the duration of whole-block verification or recovery from a failed worker. + +`StreamingExecution.VerificationWait` records wall time spent joining groups, +including failed joins and discarded streams. It overlaps `GroupJoinAssembly` for +successful groups; do not add them together or interpret it as cryptographic CPU +cost. The 100 ms limit is a conservative fallback budget, not a measured optimum. + + +### Runtime pooling default + +Omitting `tuning.use_pool` now preserves the CLI default (`true`), instead of +silently disabling VM pooling through the config reader's zero value. Explicit +TOML `false` remains supported, and an explicitly supplied CLI flag takes +precedence. This is a runtime behavior change for previously minimal configs; +pooled memory is cleared before reuse. The flag remains the single default source. diff --git a/docs/turbine-relay-buffers.md b/docs/turbine-relay-buffers.md new file mode 100644 index 000000000..3ee75f3f0 --- /dev/null +++ b/docs/turbine-relay-buffers.md @@ -0,0 +1,60 @@ +# Turbine relay buffer ownership + +Each retransmit worker reuses a private peer-result slice. Weighted shuffle, +tree placement, address filtering and fanout are unchanged. The exported +`RetransmitPeers` API still returns independently allocated result storage. +Workers clear their entire scratch slice after each send so it cannot retain +addresses from an old cluster snapshot. + +`SubmitFrom` copies a packet into exclusively owned storage before returning +to its caller. Canonical packets use a fixed-size, GC-reclaimable `sync.Pool`; +oversized inputs retain the independent allocation path. Queue admission +transfers ownership to the worker. The worker returns storage only after all +synchronous sends and retries finish, including error and no-peer paths. +Queue rejection returns storage immediately. Shutdown closes admission under +a short lock, joins workers, then drains remaining copies. The channel stays +open for concurrent submitters, which cannot enqueue after admission closes. + +`packetBatchSender.Send` borrows both packet bytes and peer addresses only +until return, even on partial sends or errors. Implementations that retain +either must copy them. A pooled packet must never escape this lifetime. +The admission lock is not held during authentication, routing or socket I/O. + +## Measurement + +`BenchmarkRetransmitPipeline` measures deduplication, packet copying, queue +handoff, weighted routing and dispatch to a mock sender. Run with: + +```sh +GOMAXPROCS=2 go test ./pkg/turbine -run '^$' \ + -bench '^BenchmarkRetransmitPipeline$' -benchmem -benchtime=20000x -count=3 +``` + +On a Ryzen 7 9700X (Zen 5), Go 1.26.4, one producer and one relay worker, +the incremental comparison against the relay implementation at +`c40ac9e8ca8aa09a9f2a8759a231199b0f90346e` was: + +| Gossip contacts | Before ns/shred | After ns/shred | Before B/shred | After B/shred | Allocations before → after | +|---|---:|---:|---:|---:|---:| +| 90 | 3,906 | 3,761 | 3,784 | ~729 | 8 → 6 | +| 512 | 23,789 | 22,056 | 3,784 | ~729 | 8 → 6 | + +Times are medians of six samples per version, in baseline/candidate/candidate/ +baseline order, three samples per round. Both versions use the same benchmark +and combined validator source; only the two relay production files change. +Each iteration submits a distinct 1,203-byte data shred with cached topology; +the benchmark checks that none are dropped and waits for workers to finish. +The fixture includes staked and zero-stake peers. Contact count does not mean +that every shred has that many recipients: tree position determines forwarding. + +Allocation fell about 81%. The 4–7% median timing improvement is modest and +noisy on the shared validator host; one candidate sample was slower than every +baseline sample in its case. These are amortized in-memory pipeline times, +not network delivery latency or live FAST-score gains. Socket syscalls, parent +authentication and retransmitter signing are excluded from this fixture. + +Routing tests compare against a full weighted permutation, including both +shuffle modes and missing/unroutable contacts. Ownership tests exercise caller +buffer reuse, queue pressure, send retries, concurrent routing and shutdown. +The full Turbine race suite passed locally and natively; native vet and the +combined validator build passed. Historical run artifacts are linked from [the evidence archive](streaming-preparation-evidence.md). diff --git a/docs/vote-delivery-persistence-evidence.md b/docs/vote-delivery-persistence-evidence.md new file mode 100644 index 000000000..1ae1f5cfa --- /dev/null +++ b/docs/vote-delivery-persistence-evidence.md @@ -0,0 +1,14 @@ +# Vote Delivery Persistence: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/54b233ff0e27fb929644f7d53bf4a699cb590cd8) +(tag `review-evidence-20260916-vote-delivery-persistence`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/54b233ff0e27fb929644f7d53bf4a699cb590cd8/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/votor-peer-isolation.md b/docs/votor-peer-isolation.md new file mode 100644 index 000000000..061b0f80a --- /dev/null +++ b/docs/votor-peer-isolation.md @@ -0,0 +1,99 @@ +# Votor outbound peer isolation + +A stalled QUIC peer could previously occupy all 32 shared send/connect workers. +quic-go's `SendDatagram` blocks when its 32-frame connection queue is full. +Repeated jobs for that peer could therefore prevent votes reaching healthy +peers, even while the broadcaster reported zero queue drops. + +Each authenticated connection now owns one sender and a FIFO queue of at most +256 encoded messages. The encoded payload is immutable and shared across peer +queues. The existing bounded worker pool handles connection attempts only; +`VotorBroadcasterConfig.Workers` controls that pool. A blocked connection cannot +consume another peer's sender or a connection worker. + +A watchdog checks every 100 ms whether the active datagram's peer-queue wait +plus its current `SendDatagram` duration has reached one second. Occasional QUIC +PTO probes can free queue entries without proving delivery; they no longer +restart this budget for an old backlog. Dequeue also retires a connection before +feeding an already-one-second-old entry into QUIC, even if sends keep completing +between watchdog ticks. The effective active-send bound is one second from +fanout enqueue, plus up to one watchdog interval and runtime scheduling delay. + +Retirement closes that connection, wakes a blocked sender, discards its queued +copies and requests a bounded reconnect. This is an operational limit on local +queueing, not a consensus validity deadline or a remote-delivery guarantee. It +adds no per-send timer or goroutine. An idle connection is not expired merely +because its previous send had a long queue delay. + +## Failure and ordering semantics + +- The global `Enqueue` contract is unchanged: rejecting a newly signed message + from the global queue returns an error to the voting engine. +- Per-peer fanout remains best effort. A full peer queue rejects that peer's + newest copy, increments both `PeerQueueDrops` and the existing + `MessagesDropped` counter, and continues sending to other peers. +- Messages queued on a failed, removed or replaced connection are discarded and + counted in `PeerQueueDiscarded`. They are not transferred to a new connection + or address. A datagram expired at dequeue is counted in `PeerQueueDiscarded`; + a failed `SendDatagram` call is counted in `PeerSendErrors`. The watchdog and + dequeue path claim retirement under the sender mutex, counting one timeout. +- Every sender exit requests a bounded, deduplicated reconnect. This covers + both a remote-close notification and closure detected while dequeuing. Peer + departure, shutdown and a healthy replacement suppress obsolete requests; + normal remote closure no longer relies on the periodic reconciliation tick. +- Messages for disconnected peers increment `PeerSendsSkipped`, as before. + This change adds no automatic application-level retransmission. +- A single sender preserves local enqueue order within its connection. QUIC + datagrams themselves still do not guarantee arrival or ordering. +- Shutdown cancels connection attempts, closes the connections outside the + broadcaster mutex, drains their queues, and waits for sender exit. + +Signing decisions, persisted vote history, durable slot reservations, and +certificate validation are unchanged. + +## Observability + +The voting log adds `peer_queue_drops`, `peer_queue_discarded`, +`peer_send_timeouts`, and `peer_queue_max_delay`. These counters/high-water marks +survive reconnects. `PeerQueueMaxDelay` measures time from fanout enqueue to +sender dequeue, including entries retired before reaching QUIC; it does not +include time blocked inside `SendDatagram`. + +Voting snapshots also expose `broadcast_peer_queues` with each current peer's +identity, address, queue depth, time in the active send, last/max queue delay, +and queue drops. Durations in the JSON snapshot use nanoseconds. Per-connection +statistics disappear when that connection is replaced; totals remain available. +Successful `PeerSends` only means quic-go accepted the datagram, not that a +remote validator received it. + +## Regression coverage + +`TestVotorBroadcasterIsolatesBlockedPeer` establishes two real loopback QUIC +connections, then blackholes one UDP path. It observes the blocked +`datagramQueue.Add` stack and requires a healthy-peer vote to arrive within +250 ms while the failed connection is still open. It also covers peer queue +overflow, watchdog reconnection, fresh traffic after reconnect, shutdown with a +blocked sender, peer departure, and replacement of a blocked peer's address. +Its connection-retirement deadline is measured from the blackhole, with explicit +scheduler slack; the healthy-peer latency assertion remains 250 ms. + +Deterministic regressions reproduce a fresh send following an old queue wait, +remote closure while idle and while dequeuing, and an already-aged queue entry. +The reconnect fixture has no reconciliation loop, so a timer cannot hide a missed +reconnect trigger. The pre-fix failures and fresh validation are retained under +[historical evidence](https://github.com/Overclock-Validator/mithril/blob/54b233ff0e27fb929644f7d53bf4a699cb590cd8/docs/results/review-fixes/2026-09-15). + +The signature-verification config template and its tests now belong to the +streaming branch, which reads those settings. This standalone voting branch +retains its base branch's supported `tuning.sigverify_backend` template. + +```sh +go test -race ./pkg/alpenglow ./pkg/consensus -count=1 -timeout=180s +go vet ./pkg/alpenglow ./pkg/consensus +go build ./cmd/mithril +``` + +This is a transport regression, not a prediction of FAST score improvement. +The live validator needs a separately validated integrated build and matching +probe addresses before deployment; the PR checkout does not include every +change in the currently enrolled validator binary. diff --git a/go.mod b/go.mod index 4607b7603..2463c0b17 100644 --- a/go.mod +++ b/go.mod @@ -5,7 +5,7 @@ go 1.26.4 replace github.com/gagliardetto/binary => github.com/palmerlao/binary v0.0.0-20250617062159-3054b4d33aed require ( - github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260726222623-da0d045dae9d + github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260730051143-c265ee966713 github.com/cespare/xxhash/v2 v2.3.0 github.com/charmbracelet/bubbles v0.21.1-0.20250623103423-23b8fd6302d7 github.com/charmbracelet/bubbletea v1.3.10 diff --git a/go.sum b/go.sum index 069571ee5..086dc43c2 100644 --- a/go.sum +++ b/go.sum @@ -12,8 +12,8 @@ github.com/Overclock-Validator/crypto v0.0.0-20250307094320-aaf52fac5261 h1:Y715 github.com/Overclock-Validator/crypto v0.0.0-20250307094320-aaf52fac5261/go.mod h1:ZhRHOaVg8I1gg0VK4wmqOQPnlgPgKFT9McZ+TCW/hBA= github.com/Overclock-Validator/gnark-crypto v0.0.0-20250309203346-2a67ed08a105 h1:mP6FWHZ8ddcmbE8UTrVVI2Mi2c24aqX/8p12Vn6zokQ= github.com/Overclock-Validator/gnark-crypto v0.0.0-20250309203346-2a67ed08a105/go.mod h1:Poczuq3dbt+CwyTKgOjGaEwJOMP7YxQobF7QhgNcguk= -github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260726222623-da0d045dae9d h1:ipaL+9MHKeI8QIfWneId0VLa+STLRM1e6MnvZhQyhPU= -github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260726222623-da0d045dae9d/go.mod h1:B7/xqV/5NtGJa8OlZAa9TRMHgeIE+VEJNiPzMP4FrIg= +github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260730051143-c265ee966713 h1:nRAD+snanlR/sX4W5rkrxr6+sYBZ3Hl2xZQ6HCpM0Hc= +github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260730051143-c265ee966713/go.mod h1:B7/xqV/5NtGJa8OlZAa9TRMHgeIE+VEJNiPzMP4FrIg= github.com/Overclock-Validator/solana-snapshot-finder-go v0.0.0-20260223201452-d8363b514fc0 h1:elgavEQb8l7Zn3gS3Y+2/98PlUylOWdlM3V1VumQ7mA= github.com/Overclock-Validator/solana-snapshot-finder-go v0.0.0-20260223201452-d8363b514fc0/go.mod h1:XbqbvMA2NKeosY0w3WdBOpCg2eYJesBjfE4cNt9HSE8= github.com/Overclock-Validator/wide v0.0.0-20250221123529-f80959d02044 h1:ph9gnWIY116AWT/iCfXoPe9/cn2aWx2uJBuLdf/LyEE= diff --git a/pkg/accounts/mem_accounts.go b/pkg/accounts/mem_accounts.go index 4158be553..630ffb420 100644 --- a/pkg/accounts/mem_accounts.go +++ b/pkg/accounts/mem_accounts.go @@ -1,7 +1,6 @@ package accounts import ( - "fmt" "sync" "github.com/Overclock-Validator/mithril/pkg/base58" @@ -13,6 +12,16 @@ type MemAccounts struct { mu *sync.RWMutex } +// A miss is an ordinary step when falling back to parent accounts. Defer the +// diagnostic encoding until it is needed, and copy the key so callers can reuse it. +type missingMemAccountError struct { + key [32]byte +} + +func (e *missingMemAccountError) Error() string { + return "no such account " + base58.Encode(e.key[:]) + " found" +} + func NewMemAccounts() MemAccounts { return MemAccounts{ Map: make(map[[32]byte]*Account), @@ -32,7 +41,7 @@ func (m MemAccounts) GetAccount(pubkey *[32]byte) (*Account, error) { defer m.mu.RUnlock() acct, ok := m.Map[*pubkey] if !ok { - return nil, fmt.Errorf("no such account %s found", base58.Encode(pubkey[:])) + return nil, &missingMemAccountError{key: *pubkey} } return acct, nil } @@ -40,7 +49,7 @@ func (m MemAccounts) GetAccount(pubkey *[32]byte) (*Account, error) { func (m MemAccounts) GetAccountWithoutLock(pubkey solana.PublicKey) (*Account, error) { acct, ok := m.Map[pubkey] if !ok { - return nil, fmt.Errorf("no such account %s found", base58.Encode(pubkey[:])) + return nil, &missingMemAccountError{key: pubkey} } return acct, nil } diff --git a/pkg/accounts/mem_accounts_test.go b/pkg/accounts/mem_accounts_test.go index bd55cb218..4a614dd9f 100644 --- a/pkg/accounts/mem_accounts_test.go +++ b/pkg/accounts/mem_accounts_test.go @@ -3,8 +3,23 @@ package accounts import ( "testing" "time" + + "github.com/gagliardetto/solana-go" ) +func TestMemAccountMissingErrorRetainsLookupKey(t *testing.T) { + mem := NewMemAccounts() + key := [32]byte{} + _, lockedErr := mem.GetAccount(&key) + _, unlockedErr := mem.GetAccountWithoutLock(solana.PublicKey(key)) + key[0] = 99 // Lookup callers may reuse their key storage before reporting an error. + for _, err := range []error{lockedErr, unlockedErr} { + if err == nil || err.Error() != "no such account 11111111111111111111111111111111 found" { + t.Fatalf("missing error lost its original key: %v", err) + } + } +} + func TestMemAccountsReadsAreConcurrent(t *testing.T) { mem := NewMemAccounts() var key [32]byte diff --git a/pkg/accounts/overlay_test.go b/pkg/accounts/overlay_test.go index f40a26dfc..682e47217 100644 --- a/pkg/accounts/overlay_test.go +++ b/pkg/accounts/overlay_test.go @@ -411,3 +411,42 @@ func TestOverlayDeltaAccountsIncludesOverride(t *testing.T) { assert.Equal(t, pk(1), delta[0].Key) assert.Equal(t, uint64(99), delta[0].Lamports) } + +func TestWorkingSetPromotionChunkBoundaries(t *testing.T) { + w := NewWorkingSet() + for _, slot := range []uint64{5, 7, 9, 11} { + w.Add(slot, []*Account{uoAcct(1, slot), uoAcct(2, slot+100)}) + } + for _, tc := range []struct { + through uint64 + limit int + partial bool + slots []uint64 + }{ + {4, 2, true, nil}, {5, 2, false, nil}, {7, 2, false, []uint64{5, 7}}, + {11, 2, false, []uint64{5, 7}}, {9, 4, true, []uint64{5, 7, 9}}, + {9, 4, false, nil}, {11, 0, true, nil}, {11, -1, false, nil}, + } { + got := w.PromotionChunk(tc.through, tc.limit, tc.partial) + var slots []uint64 + for _, sd := range got { + slots = append(slots, sd.Slot) + require.Len(t, sd.Delta, 2) + for _, acct := range sd.Delta { + require.True(t, acct.Lamports == sd.Slot || acct.Lamports == sd.Slot+100) + } + } + require.Equal(t, tc.slots, slots) + } + // Preparing a job leaves the live suffix intact. Once the caller commits + // and promotes a prefix, the next chunk must start at the surviving slot. + require.Equal(t, 4, w.HeldSlots()) + w.PromotePrefix(7) + chunk := w.PromotionChunk(11, 2, false) + require.Equal(t, []uint64{9, 11}, []uint64{chunk[0].Slot, chunk[1].Slot}) + require.Zero(t, testing.AllocsPerRun(100, func() { + if w.PromotionChunk(9, 2, false) != nil { + panic("partial chunk escaped") + } + })) +} diff --git a/pkg/accounts/working_set.go b/pkg/accounts/working_set.go index d88725f24..083a4405d 100644 --- a/pkg/accounts/working_set.go +++ b/pkg/accounts/working_set.go @@ -134,17 +134,37 @@ func (w *WorkingSet) PromotionPrefix(through uint64) []SlotDelta { w.mu.RLock() defer w.mu.RUnlock() - var batch []SlotDelta - for _, slot := range w.order { // ascending - if slot > through { - break - } + return w.promotionChunkLocked(through, len(w.order), true) +} + +// PromotionChunk returns at most maxSlots oldest held slots through the caller's +// verified promotion bound. Unless allowPartial is set, an incomplete chunk +// returns nil before allocating or collecting account writes. Selection and +// collection share one read lock, so pruning cannot change the selected prefix. +// This only prepares borrowed account pointers; it does not commit, prune, or +// advance durability. Callers still own finality checks and durable commit order. +func (w *WorkingSet) PromotionChunk(through uint64, maxSlots int, allowPartial bool) []SlotDelta { + w.mu.RLock() + defer w.mu.RUnlock() + return w.promotionChunkLocked(through, maxSlots, allowPartial) +} + +func (w *WorkingSet) promotionChunkLocked(through uint64, maxSlots int, allowPartial bool) []SlotDelta { + count := 0 + for count < len(w.order) && count < maxSlots && w.order[count] <= through { + count++ + } + if count == 0 || (!allowPartial && count < maxSlots) { + return nil + } + batch := make([]SlotDelta, count) + for i, slot := range w.order[:count] { layer := w.bySlot[slot] delta := make([]*Account, 0, len(layer.writes)) for _, a := range layer.writes { delta = append(delta, a) } - batch = append(batch, SlotDelta{Slot: slot, Delta: delta}) + batch[i] = SlotDelta{Slot: slot, Delta: delta} } return batch } diff --git a/pkg/accountsdb/accountsdb.go b/pkg/accountsdb/accountsdb.go index 49714f731..acc1666e5 100644 --- a/pkg/accountsdb/accountsdb.go +++ b/pkg/accountsdb/accountsdb.go @@ -9,6 +9,7 @@ import ( "errors" "fmt" "log" + "maps" "os" "path/filepath" "runtime/trace" @@ -18,6 +19,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/accounts" "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/features" "github.com/Overclock-Validator/mithril/pkg/mlog" "github.com/Overclock-Validator/mithril/pkg/sbpf" "github.com/cockroachdb/pebble" @@ -251,6 +253,24 @@ func (accountsDb *AccountsDb) InitCaches() { type ProgramCacheEntry struct { Program *sbpf.Program DeploymentSlot uint64 + // Executables are reusable across banks only for identical source and loader + // features. Slot alone is not a version: competing forks can deploy at the + // same slot. Fields are immutable after cache publication. + sourceBytes []byte + sourceBound bool + sourceFeatures features.Features +} + +// BindSource must be called before publishing the entry; published bindings +// must never be mutated, including when another bank replaces the cache key. +func (entry *ProgramCacheEntry) BindSource(source []byte, f *features.Features) { + entry.sourceBytes = bytes.Clone(source) + entry.sourceBound = true + entry.sourceFeatures = *f.Clone() +} + +func (entry *ProgramCacheEntry) MatchesSource(source []byte, f *features.Features) bool { + return entry != nil && entry.sourceBound && bytes.Equal(entry.sourceBytes, source) && maps.Equal(entry.sourceFeatures, *f) } func programCacheCapacityUnits() int { @@ -287,7 +307,7 @@ func (entry *ProgramCacheEntry) CostUnits() uint32 { if entry == nil || entry.Program == nil { return 1 } - bytes := entry.Program.MemoryBytes() + bytes := entry.Program.MemoryBytes() + uint64(len(entry.sourceBytes)) units := (bytes + programCacheCostUnitBytes - 1) / programCacheCostUnitBytes if units == 0 { return 1 diff --git a/pkg/accountsdb/program_cache_version_test.go b/pkg/accountsdb/program_cache_version_test.go new file mode 100644 index 000000000..1d9d962a0 --- /dev/null +++ b/pkg/accountsdb/program_cache_version_test.go @@ -0,0 +1,31 @@ +package accountsdb + +import ( + "github.com/Overclock-Validator/mithril/pkg/features" + "testing" +) + +func TestProgramCacheSourceBinding(t *testing.T) { + f := features.NewFeaturesDefault() + source := []byte{1, 2, 3} + entry := &ProgramCacheEntry{DeploymentSlot: 100} + if entry.MatchesSource(source, f) { + t.Fatal("unbound entry matched") + } + entry.BindSource(source, f) + if !entry.MatchesSource(source, f) { + t.Fatal("identical source missed") + } + source[0]++ + if entry.MatchesSource(source, f) { + t.Fatal("same-slot fork source matched") + } + source[0]-- + f.EnableFeature(features.VirtualAddressSpaceAdjustments, 0) + if entry.MatchesSource(source, f) { + t.Fatal("different feature environment matched") + } + if !entry.MatchesSource(source, features.NewFeaturesDefault()) { + t.Fatal("binding retained mutable feature map") + } +} diff --git a/pkg/alpenglow/broadcaster.go b/pkg/alpenglow/broadcaster.go index bd8e3ee7c..557ea997d 100644 --- a/pkg/alpenglow/broadcaster.go +++ b/pkg/alpenglow/broadcaster.go @@ -18,7 +18,7 @@ import ( const ( defaultVotorBroadcastQueue = 1024 - defaultVotorSendWorkers = 32 + defaultVotorConnectWorkers = 32 defaultVotorPeerJobQueue = 16384 defaultVotorPeerRefreshInterval = time.Second ) @@ -35,7 +35,8 @@ type VotorBroadcasterConfig struct { ShredVersion uint16 Peers VotorPeerSource QueueSize int - Workers int + // Workers bounds concurrent connection attempts. Sends are isolated per connection. + Workers int } type VotorBroadcasterStats struct { @@ -50,28 +51,25 @@ type VotorBroadcasterStats struct { ConnectionAttempts uint64 ConnectionErrors uint64 ConnectionJobsDropped uint64 + PeerQueueDrops uint64 + PeerQueueDiscarded uint64 + PeerSendTimeouts uint64 + PeerQueueMaxDelay time.Duration + PeerQueues []VotorPeerQueueStats LastPeerSendError string LastPeerSendErrorAt time.Time LastConnectionError string LastConnectionErrorAt time.Time } -type votorPeerJobKind uint8 - -const ( - votorPeerJobConnect votorPeerJobKind = iota - votorPeerJobSend -) - type votorPeerJob struct { - kind votorPeerJobKind - peer VotorPeer - payload []byte + peer VotorPeer } type votorConnection struct { - addr string - conn *quic.Conn + addr string + conn *quic.Conn + sender *votorPeerSender } type votorDial struct { @@ -109,6 +107,10 @@ type VotorBroadcaster struct { connectionAttempts atomic.Uint64 connectionErrors atomic.Uint64 connectionJobsDropped atomic.Uint64 + peerQueueDrops atomic.Uint64 + peerQueueDiscarded atomic.Uint64 + peerSendTimeouts atomic.Uint64 + peerQueueMaxDelay atomic.Int64 } func NewVotorBroadcaster(cfg VotorBroadcasterConfig) (*VotorBroadcaster, error) { @@ -122,7 +124,7 @@ func NewVotorBroadcaster(cfg VotorBroadcasterConfig) (*VotorBroadcaster, error) cfg.QueueSize = defaultVotorBroadcastQueue } if cfg.Workers <= 0 { - cfg.Workers = defaultVotorSendWorkers + cfg.Workers = defaultVotorConnectWorkers } certificate, err := newVotorQUICCertificate(cfg.Identity) if err != nil { @@ -151,7 +153,7 @@ func NewVotorBroadcaster(cfg VotorBroadcasterConfig) (*VotorBroadcaster, error) } b.wg.Add(2 + cfg.Workers) for range cfg.Workers { - go b.sendLoop() + go b.connectLoop() } go b.broadcastLoop() // Populate the desired set and queue the first bounded preconnects before @@ -199,71 +201,43 @@ func (b *VotorBroadcaster) broadcastLoop() { b.recordSendError(VotorPeer{}, fmt.Errorf("encode Votor message: %w", err)) continue } - peers, skipped := b.connectedPeers() + senders, skipped := b.connectedSenders() b.sendsSkipped.Add(uint64(skipped)) - for _, peer := range peers { - job := votorPeerJob{kind: votorPeerJobSend, peer: peer, payload: payload} - select { - case b.jobs <- job: - case <-b.done: - return - default: - b.dropped.Add(1) - } + job := votorDatagram{payload: payload, queuedAt: time.Now()} + for _, sender := range senders { + sender.enqueue(job) } } } } -func (b *VotorBroadcaster) sendLoop() { +// Connection attempts never occupy a peer's sender or delay connected peers. +func (b *VotorBroadcaster) connectLoop() { defer b.wg.Done() for { select { case <-b.done: return case job := <-b.jobs: - switch job.kind { - case votorPeerJobConnect: - b.connectPeer(job.peer.Identity) - case votorPeerJobSend: - if err := b.send(job.peer, job.payload); err != nil { - b.recordSendError(job.peer, err) - } else { - b.sends.Add(1) - } - } + b.connectPeer(job.peer.Identity) } } } -func (b *VotorBroadcaster) send(peer VotorPeer, payload []byte) error { - conn, ok := b.establishedConnection(peer) - if !ok { - b.queueConnect(peer.Identity) - return fmt.Errorf("send Votor datagram to %s (%s): no established connection", peer.Identity, peer.Addr) - } - err := conn.SendDatagram(payload) - if err == nil { - return nil - } - var tooLarge *quic.DatagramTooLargeError - if !errors.As(err, &tooLarge) { - b.dropConnection(peer.Identity, conn) - b.queueConnect(peer.Identity) - } - return fmt.Errorf("send Votor datagram to %s (%s): %w", peer.Identity, peer.Addr, err) -} - func (b *VotorBroadcaster) peerReconcileLoop() { defer b.wg.Done() ticker := time.NewTicker(defaultVotorPeerRefreshInterval) defer ticker.Stop() + watchdog := time.NewTicker(votorSendWatchInterval) + defer watchdog.Stop() for { select { case <-b.done: return case <-ticker.C: b.reconcilePeers() + case now := <-watchdog.C: + b.expirePeerSends(now) } } } @@ -305,9 +279,9 @@ func (b *VotorBroadcaster) reconcilePeers() { } } -func (b *VotorBroadcaster) connectedPeers() ([]VotorPeer, int) { +func (b *VotorBroadcaster) connectedSenders() ([]*votorPeerSender, int) { b.connMu.Lock() - peers := make([]VotorPeer, 0, len(b.desired)) + peers := make([]*votorPeerSender, 0, len(b.desired)) skipped := 0 for identity, peer := range b.desired { existing, connected := b.conns[identity] @@ -315,8 +289,7 @@ func (b *VotorBroadcaster) connectedPeers() ([]VotorPeer, int) { skipped++ continue } - peer.Addr = cloneUDPAddr(peer.Addr) - peers = append(peers, peer) + peers = append(peers, existing.sender) } b.connMu.Unlock() return peers, skipped @@ -349,7 +322,7 @@ func (b *VotorBroadcaster) queueConnectLocked(identity solana.PublicKey) { if _, queued := b.connectQueued[identity]; queued || b.dialing[identity] != nil { return } - job := votorPeerJob{kind: votorPeerJobConnect, peer: peer} + job := votorPeerJob{peer: peer} select { case b.jobs <- job: b.connectQueued[identity] = struct{}{} @@ -476,7 +449,12 @@ func (b *VotorBroadcaster) connection(peer VotorPeer) (*quic.Conn, error) { return existing.conn, nil } stale := b.conns[peer.Identity].conn - b.conns[peer.Identity] = votorConnection{addr: addr, conn: conn} + sender := &votorPeerSender{b: b, peer: peer, conn: conn, queue: make(chan votorDatagram, defaultVotorPeerSendQueue), done: make(chan struct{})} + b.conns[peer.Identity] = votorConnection{addr: addr, conn: conn, sender: sender} + // Close takes connMu before waiting, so no sender can be added after it + // observes the closed flag and drains the connection set. + b.wg.Add(1) + go sender.run() b.connMu.Unlock() if stale != nil { _ = stale.CloseWithError(0, "Votor peer address changed") @@ -499,12 +477,14 @@ func (b *VotorBroadcaster) Stats() VotorBroadcasterStats { } b.connMu.Lock() connections := 0 + peerQueues := make([]VotorPeerQueueStats, 0, len(b.conns)) for identity, existing := range b.conns { if existing.conn.Context().Err() != nil { delete(b.conns, identity) continue } connections++ + peerQueues = append(peerQueues, existing.sender.stats()) } desiredPeers := len(b.desired) pendingConnections := len(b.connectQueued) @@ -525,6 +505,11 @@ func (b *VotorBroadcaster) Stats() VotorBroadcasterStats { ConnectionAttempts: b.connectionAttempts.Load(), ConnectionErrors: b.connectionErrors.Load(), ConnectionJobsDropped: b.connectionJobsDropped.Load(), + PeerQueueDrops: b.peerQueueDrops.Load(), + PeerQueueDiscarded: b.peerQueueDiscarded.Load(), + PeerSendTimeouts: b.peerSendTimeouts.Load(), + PeerQueueMaxDelay: time.Duration(b.peerQueueMaxDelay.Load()), + PeerQueues: peerQueues, LastPeerSendError: lastSendError, LastPeerSendErrorAt: lastSendErrorAt, LastConnectionError: lastConnectionError, @@ -542,11 +527,15 @@ func (b *VotorBroadcaster) Close() error { close(b.done) b.connMu.Lock() clear(b.desired) + stale := make([]*quic.Conn, 0, len(b.conns)) for identity, existing := range b.conns { - _ = existing.conn.CloseWithError(0, "Votor broadcaster closed") + stale = append(stale, existing.conn) delete(b.conns, identity) } b.connMu.Unlock() + for _, conn := range stale { + _ = conn.CloseWithError(0, "Votor broadcaster closed") + } b.wg.Wait() }) return nil diff --git a/pkg/alpenglow/certpool.go b/pkg/alpenglow/certpool.go index 60b1882e3..4383898d3 100644 --- a/pkg/alpenglow/certpool.go +++ b/pkg/alpenglow/certpool.go @@ -5,8 +5,11 @@ import ( "crypto/sha256" "fmt" "math/big" + "runtime" "sync" + "sync/atomic" + "github.com/Overclock-Validator/gnark-crypto/ecc" bls12381 "github.com/Overclock-Validator/gnark-crypto/ecc/bls12-381" blsfr "github.com/Overclock-Validator/gnark-crypto/ecc/bls12-381/fr" "github.com/Overclock-Validator/mithril/pkg/mlog" @@ -31,7 +34,7 @@ const maxUnverifiedCandidatesPerRank = 2 // facts. Ingest (AddVote) only shape-checks, bounds buffering, and parks the // vote — it NEVER mutates dedupe, equivocation, disjointness, or tally state // that assembly or fork choice depends on. Those mutate only AFTER the BLS -// signature verifies (foldTallyLocked). This prevents a bogus vote for a +// signature verifies (verifyAndFoldTallyWithLockReleased). This prevents a bogus vote for a // victim rank from suppressing that validator's real vote (dedupe poisoning) // or forging equivocation evidence against an honest validator. // @@ -109,14 +112,12 @@ type tally struct { pending map[uint16]map[[sha256.Size]byte]VoteMessage // unverified candidates, by rank and signature verified map[uint16]struct{} // ranks folded into the aggregate aggSig bls12381.G2Affine // sum of verified signatures - aggPub bls12381.G1Affine // sum of verified pubkeys stake uint64 // verified stake } func newTally() *tally { t := &tally{pending: make(map[uint16]map[[sha256.Size]byte]VoteMessage), verified: make(map[uint16]struct{})} t.aggSig.SetInfinity() - t.aggPub.SetInfinity() return t } @@ -136,7 +137,14 @@ type tallyKey struct { } type poolSlot struct { - tallies map[tallyKey]*tally + // One caller owns folding for a slot while other callers may buffer votes. + // All fields, like the maps below, are protected by CertPool.mu. + processing bool + // Queued and active reward flushes take precedence over new arrivals/owners. + // A count preserves priority when multiple footer builders overlap. + flushWaiting int + dirty bool + tallies map[tallyKey]*tally // verifiedHash tracks the block hashes a rank has cast VERIFIED votes for, // per (rank, type), for equivocation/vote-budget enforcement. Populated only // after signature verification — never from raw ingest — so a bogus vote can @@ -166,16 +174,20 @@ type CertPool struct { // end and must not make downstream consensus decisions itself. verifiedVoteSink func(VerifiedVote) - mu sync.Mutex - epochForSlot func(slot uint64) uint64 - slots map[uint64]*poolSlot - emitted map[CertificateKey]struct{} - floor uint64 - liveSlot uint64 // trusted replay/observed watermark (NOT advanced by raw votes) - highestSlot uint64 // observability only: highest vote slot seen - totalPending int - equivocation []EquivocationEvidence - snap CertPoolSnapshot + mu sync.Mutex + workCond *sync.Cond + verificationMu sync.Mutex // Keep expensive batch concurrency bounded at one. + batchesVerified atomic.Uint64 + epochGeneration uint64 + epochForSlot func(slot uint64) uint64 + slots map[uint64]*poolSlot + emitted map[CertificateKey]struct{} + floor uint64 + liveSlot atomic.Uint64 // trusted replay/observed watermark (NOT advanced by raw votes) + highestSlot uint64 // observability only: highest vote slot seen + totalPending int + equivocation []EquivocationEvidence + snap CertPoolSnapshot publicationMu sync.Mutex publicationCond *sync.Cond @@ -214,6 +226,7 @@ func NewCertPool(cfg CertPoolConfig, verifier *CertificateVerifier, emit func(Ce publicationCompleted: make(map[uint64]struct{}), } p.publicationCond = sync.NewCond(&p.publicationMu) + p.workCond = sync.NewCond(&p.mu) return p } @@ -229,6 +242,7 @@ func (p *CertPool) SetVerifiedVoteSink(sink func(VerifiedVote)) { func (p *CertPool) SetEpochLookup(fn func(slot uint64) uint64) { p.mu.Lock() p.epochForSlot = fn + p.epochGeneration++ p.mu.Unlock() } @@ -236,21 +250,24 @@ func (p *CertPool) SetEpochLookup(fn func(slot uint64) uint64) { // (from replay progress / observed finality — never from raw votes, so an // attacker cannot slide the window forward). Monotonic. func (p *CertPool) NoteLiveSlot(slot uint64) { - p.mu.Lock() - if slot > p.liveSlot { - p.liveSlot = slot + // Replay must not wait for BLS verification under p.mu merely to announce + // progress. Concurrent trusted updates may arrive out of order. + for current := p.liveSlot.Load(); slot > current; current = p.liveSlot.Load() { + if p.liveSlot.CompareAndSwap(current, slot) { + return + } } - p.mu.Unlock() } // windowAnchorLocked is the trusted upper anchor of the live vote window: the // higher of the finalized floor and the replay-observed live slot. It is NOT // derived from raw votes, so ingest cannot advance it. func (p *CertPool) windowAnchorLocked() uint64 { - if p.floor > p.liveSlot { + liveSlot := p.liveSlot.Load() + if p.floor > liveSlot { return p.floor } - return p.liveSlot + return liveSlot } // setForSlotLocked resolves the validator set covering slot. Returns nil (votes @@ -271,45 +288,60 @@ func (p *CertPool) setForSlotLocked(slot uint64) *ValidatorSet { // AddVote ingests one raw (unverified) votor vote. It ONLY shape-checks, bounds // buffering, and parks the vote in its (type, hash) tally. It does NOT touch // dedupe/equivocation/disjointness state — those mutate only after the vote's -// signature verifies (foldTallyLocked). A malformed vote, a vote outside the +// signature verifies (verifyAndFoldTallyWithLockReleased). A malformed vote, a vote outside the // trusted slot window, or a vote past a memory bound is dropped. func (p *CertPool) AddVote(msg VoteMessage) { if msg.Vote.ValidateBasic() != nil || len(msg.Signature) != BLSSignatureSize { return } slot := msg.Vote.Slot + sigKey := sha256.Sum256(msg.Signature) - var emits []Certificate - var verified []VerifiedVote p.mu.Lock() - if slot <= p.floor { - p.snap.VotesRejected++ - p.mu.Unlock() - return - } - // Window anchored to the TRUSTED watermark (floor / replay-observed), not to - // the highest vote slot seen — otherwise an attacker could slide it forward - // vote by vote and retain arbitrarily many future slots. - if anchor := p.windowAnchorLocked(); anchor > 0 && slot > anchor+p.cfg.MaxSlotsAhead { - p.snap.VotesRejected++ - p.mu.Unlock() - return - } - - ps := p.slots[slot] - if ps == nil { - // Hard global bound on retained slots (independent of the window). - if len(p.slots) >= p.cfg.MaxLiveSlots && !p.evictFartherFutureSlotLocked(slot) { + var ps *poolSlot + for { + if slot <= p.floor { + p.snap.VotesRejected++ + p.mu.Unlock() + return + } + // Window anchored to the TRUSTED watermark (floor / replay-observed), not to + // the highest vote slot seen — otherwise an attacker could slide it forward + // vote by vote and retain arbitrarily many future slots. + if anchor := p.windowAnchorLocked(); anchor > 0 && slot > anchor+p.cfg.MaxSlotsAhead { p.snap.VotesRejected++ p.mu.Unlock() return } - ps = &poolSlot{ - tallies: make(map[tallyKey]*tally), - verifiedHash: make(map[voteDedupKey][]solana.Hash), - pendingByRank: make(map[uint16]int), + + ps = p.slots[slot] + if ps == nil { + // Hard global bound on retained slots (independent of the window). + if len(p.slots) >= p.cfg.MaxLiveSlots && !p.evictFartherFutureSlotLocked(slot) { + p.snap.VotesRejected++ + p.mu.Unlock() + return + } + ps = &poolSlot{ + tallies: make(map[tallyKey]*tally), + verifiedHash: make(map[voteDedupKey][]solana.Hash), + pendingByRank: make(map[uint16]int), + } + p.slots[slot] = ps } - p.slots[slot] = ps + // Ordinary arrivals can join the bounded pending maps during BLS work. + // Quota pressure and competing signatures still wait for authentication + // before admission, preserving the first-packet-poisoning protections. + // A queued reward flush also stops new arrivals from extending its drain. + if ps.flushWaiting > 0 || (ps.processing && p.admissionNeedsFoldLocked(ps, msg, sigKey)) { + p.workCond.Wait() + continue + } + break + } + owner := !ps.processing + if owner { + ps.processing = true } tk := tallyKey{Type: msg.Vote.Type, Hash: msg.Vote.BlockHash} @@ -324,7 +356,6 @@ func (p *CertPool) AddVote(msg VoteMessage) { // one. Different block hashes remain separate tallies. forceFold := false if _, done := tl.verified[msg.Rank]; !done { - sigKey := sha256.Sum256(msg.Signature) candidates := tl.pending[msg.Rank] _, duplicate := candidates[sigKey] if !duplicate && ps.pendingByRank[msg.Rank] >= p.cfg.MaxPendingVotesPerRankSlot { @@ -332,7 +363,11 @@ func (p *CertPool) AddVote(msg VoteMessage) { // the rank quota, authenticate that rank's parked candidates and free // the invalid ones before deciding whether the real vote has room. if set := p.setForSlotLocked(slot); set != nil { - verified = append(verified, p.foldPendingRankLocked(slot, ps, msg.Rank, set)...) + p.foldPendingRankLocked(slot, ps, msg.Rank, set) + } + if p.slots[slot] != ps { + p.finishSlotAndUnlock(slot, ps, nil) + return } candidates = tl.pending[msg.Rank] } @@ -342,7 +377,11 @@ func (p *CertPool) AddVote(msg VoteMessage) { atCap := ps.pendingCount >= p.cfg.MaxPendingVotesPerSlot || p.totalPending >= p.cfg.MaxPendingVotesTotal if !rejectIncoming && atCap { if set := p.setForSlotLocked(slot); set != nil { - verified = append(verified, p.foldAllPendingLocked(slot, ps, set)...) + p.foldAllPendingLocked(slot, ps, set) + } + if p.slots[slot] != ps { + p.finishSlotAndUnlock(slot, ps, nil) + return } if p.totalPending >= p.cfg.MaxPendingVotesTotal { p.evictFartherFutureSlotLocked(slot) @@ -382,90 +421,161 @@ func (p *CertPool) AddVote(msg VoteMessage) { if slot > p.highestSlot { p.highestSlot = slot } + if !owner { + ps.dirty = true + p.mu.Unlock() + return + } if forceFold { if set := p.setForSlotLocked(slot); set != nil { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) + } + } + emits := p.drainSlotLocked(slot, ps, false) + p.finishSlotAndUnlock(slot, ps, emits) +} + +func (p *CertPool) admissionNeedsFoldLocked(ps *poolSlot, msg VoteMessage, sigKey [sha256.Size]byte) bool { + tl := ps.tallies[tallyKey{Type: msg.Vote.Type, Hash: msg.Vote.BlockHash}] + if tl != nil { + if _, done := tl.verified[msg.Rank]; done { + return false + } + candidates := tl.pending[msg.Rank] + if _, duplicate := candidates[sigKey]; duplicate { + return false + } + if len(candidates) > 0 { + return true } } - // Keep verified stake fresh for the Votor fallback triggers. Only plain - // notarize/skip arrivals can change the trigger predicates. - if msg.Vote.Type == VoteTypeNotarize || msg.Vote.Type == VoteTypeSkip { - verified = append(verified, p.maybeFoldTriggersLocked(slot, ps)...) + return ps.pendingByRank[msg.Rank] >= p.cfg.MaxPendingVotesPerRankSlot || + ps.pendingCount >= p.cfg.MaxPendingVotesPerSlot || p.totalPending >= p.cfg.MaxPendingVotesTotal +} + +// waitForSlotLocked releases mu while the current owner finishes. Pruning can +// remove or replace the slot, so callers always use the returned current state. +func (p *CertPool) waitForSlotLocked(slot uint64) *poolSlot { + for { + ps := p.slots[slot] + if ps == nil || (!ps.processing && ps.flushWaiting == 0) { + return ps + } + p.workCond.Wait() } - var assembledVerified []VerifiedVote - emits, assembledVerified = p.maybeAssembleLocked(slot, ps) - verified = append(verified, assembledVerified...) - sink := p.verifiedVoteSink - publication := p.reservePublicationLocked(verified) - p.mu.Unlock() +} - p.publishVerifiedVotes(sink, verified, publication) +// drainSlotLocked catches arrivals buffered while the owner was outside mu. +// Sub-threshold votes remain lazy except when preparing a reward footer. +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) drainSlotLocked(slot uint64, ps *poolSlot, flush bool) []Certificate { + var emits []Certificate + for p.slots[slot] == ps { + ps.dirty = false + if flush { + if set := p.setForSlotLocked(slot); set != nil { + for key, tl := range ps.tallies { + if key.Type == VoteTypeSkip || key.Type == VoteTypeNotarize { + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) + } + } + } + } + p.maybeFoldTriggersLocked(slot, ps) + certs := p.maybeAssembleLocked(slot, ps) + emits = append(emits, certs...) + if !ps.dirty { + break + } + } + return emits +} + +// finishSlotAndUnlock requires p.mu held and returns with p.mu released. +// It takes the publication barrier before making the slot available to a +// flushing caller, then unlocks and emits certificates. Callers must not defer +// an unlock across this call. +func (p *CertPool) finishSlotAndUnlock(slot uint64, ps *poolSlot, emits []Certificate) uint64 { + if p.slots[slot] != ps { + emits = nil + } + p.snap.CertsEmitted += uint64(len(emits)) + target := p.publicationTargetLocked() + ps.processing = false + p.workCond.Broadcast() + p.mu.Unlock() for _, cert := range emits { p.emitCert(cert) } + return target } // OnValidatorSetInstalled retries assembly for buffered slots that resolve to // the newly-installed epoch. Requires a real slot→epoch lookup; without one no // slot can be safely attributed to the epoch, so nothing is retried. func (p *CertPool) OnValidatorSetInstalled(epoch uint64) { - var emits []Certificate - var verified []VerifiedVote p.mu.Lock() + var slots []uint64 if p.epochForSlot != nil { - for slot, ps := range p.slots { - if p.epochForSlot(slot) != epoch { - continue + for slot := range p.slots { + if p.epochForSlot(slot) == epoch { + slots = append(slots, slot) } - verified = append(verified, p.maybeFoldTriggersLocked(slot, ps)...) - certs, newlyVerified := p.maybeAssembleLocked(slot, ps) - emits = append(emits, certs...) - verified = append(verified, newlyVerified...) } } - sink := p.verifiedVoteSink - publication := p.reservePublicationLocked(verified) p.mu.Unlock() - p.publishVerifiedVotes(sink, verified, publication) - for _, cert := range emits { - p.emitCert(cert) + for _, slot := range slots { + p.mu.Lock() + ps := p.waitForSlotLocked(slot) + if ps == nil || p.epochForSlot == nil || p.epochForSlot(slot) != epoch { + p.mu.Unlock() + continue + } + ps.processing = true + emits := p.drainSlotLocked(slot, ps, false) + p.finishSlotAndUnlock(slot, ps, emits) } } // FlushRewardVotes batch-verifies every pending plain skip/notarize vote for a // reward slot. Normal consensus verification stays lazy, but block production // calls this just before building the slot+8 footer so valid below-threshold -// votes are not omitted from reward certificates. +// votes are not omitted from reward certificates. Queued flushes take precedence +// over new slot owners and arrivals; an existing owner finishes first. Pruning +// can invalidate that generation, in which case the current slot is rechecked. func (p *CertPool) FlushRewardVotes(slot uint64) { - var emits []Certificate - var verified []VerifiedVote p.mu.Lock() - ps := p.slots[slot] - set := p.setForSlotLocked(slot) - if ps != nil && set != nil { - for key, tl := range ps.tallies { - if key.Type == VoteTypeSkip || key.Type == VoteTypeNotarize { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) - } + var ps *poolSlot + for { + ps = p.slots[slot] + if ps == nil { + break + } + ps.flushWaiting++ + for p.slots[slot] == ps && ps.processing { + p.workCond.Wait() } - var assembled []VerifiedVote - emits, assembled = p.maybeAssembleLocked(slot, ps) - verified = append(verified, assembled...) + if p.slots[slot] == ps { + break + } + // Pruning or a binding change replaced this generation while waiting. + ps.flushWaiting-- + p.workCond.Broadcast() } - sink := p.verifiedVoteSink - publication := p.reservePublicationLocked(verified) - targetPublication := p.publicationTargetLocked() - p.mu.Unlock() - - p.publishVerifiedVotes(sink, verified, publication) - for _, cert := range emits { - p.emitCert(cert) + if ps == nil { + target := p.publicationTargetLocked() + p.mu.Unlock() + p.waitForPublication(target) + return } - // A vote can finish BLS verification in AddVote just before this flush takes - // the pool lock, then still be in flight to the reward builder. Wait through - // that publication sequence so a footer cannot omit an already-verified vote. - p.waitForPublication(targetPublication) + ps.processing = true + emits := p.drainSlotLocked(slot, ps, true) + ps.flushWaiting-- + target := p.finishSlotAndUnlock(slot, ps, emits) + // Includes publication by an owner that was verifying when flush arrived. + p.waitForPublication(target) } // reservePublicationLocked assigns ordering while p.mu is held. Flush can then @@ -541,6 +651,7 @@ func (p *CertPool) ObserveFloor(finalizedSlot uint64) { delete(p.emitted, key) } } + p.workCond.Broadcast() } p.mu.Unlock() } @@ -557,10 +668,11 @@ func (p *CertPool) Snapshot() CertPoolSnapshot { p.mu.Lock() defer p.mu.Unlock() snap := p.snap + snap.BatchesVerified = p.batchesVerified.Load() snap.Slots = len(p.slots) snap.Floor = p.floor snap.HighestSlot = p.highestSlot - snap.LiveSlot = p.liveSlot + snap.LiveSlot = p.liveSlot.Load() snap.PendingTotal = p.totalPending return snap } @@ -648,10 +760,15 @@ func meets(f Fraction, stake, total uint64) bool { // implementation (Agave) would. This is the ONLY fold policy: one fork-choice // behavior for observer and voting nodes alike; sub-trigger tallies still // cost nothing. -func (p *CertPool) maybeFoldTriggersLocked(slot uint64, ps *poolSlot) []VerifiedVote { +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) maybeFoldTriggersLocked(slot uint64, ps *poolSlot) { + if p.slots[slot] != ps { + return + } set := p.setForSlotLocked(slot) if set == nil { - return nil // votes stay buffered; retried on OnValidatorSetInstalled + return // votes stay buffered; retried on OnValidatorSetInstalled } total := set.TotalStake v := buildTriggerViewLocked(ps, set) @@ -686,22 +803,21 @@ func (p *CertPool) maybeFoldTriggersLocked(slot uint64, ps *poolSlot) []Verified } if !foldSkip && !foldAllNotar && len(foldNotar) == 0 { - return nil + return } - var verified []VerifiedVote for tk, tl := range ps.tallies { switch tk.Type { case VoteTypeNotarize: if foldAllNotar || foldNotar[tk.Hash] { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) } case VoteTypeSkip: if foldSkip { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) } } } - return verified + return } // VotorStakes is the verified-stake observation Votor's fallback-trigger @@ -723,12 +839,12 @@ type VotorStakes struct { func (p *CertPool) VerifiedVotorStakes(slot uint64) (VotorStakes, bool) { p.mu.Lock() defer p.mu.Unlock() + ps := p.waitForSlotLocked(slot) set := p.setForSlotLocked(slot) if set == nil { return VotorStakes{}, false } out := VotorStakes{Notarize: make(map[solana.Hash]uint64), TotalStake: set.TotalStake} - ps := p.slots[slot] if ps == nil { return out, true } @@ -804,14 +920,18 @@ func targetsForSlot(ps *poolSlot) []certTarget { // maybeAssembleLocked checks every assemblable target for the slot: folds // pending votes (batch verification) once candidate stake crosses the // threshold, and returns any newly assembled certificates for emission. -func (p *CertPool) maybeAssembleLocked(slot uint64, ps *poolSlot) ([]Certificate, []VerifiedVote) { +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) maybeAssembleLocked(slot uint64, ps *poolSlot) []Certificate { + if p.slots[slot] != ps { + return nil + } set := p.setForSlotLocked(slot) if set == nil { - return nil, nil // validator set / epoch not resolvable yet; votes stay buffered + return nil // validator set / epoch not resolvable yet; votes stay buffered } var emits []Certificate - var verified []VerifiedVote for _, target := range targetsForSlot(ps) { key := CertificateKey{Type: target.certType, Slot: slot} if target.certType.HasBlock() { @@ -846,8 +966,11 @@ func (p *CertPool) maybeAssembleLocked(slot uint64, ps *poolSlot) ([]Certificate } // Candidate stake crossed: fold pending votes (one pairing per tally). - verified = append(verified, p.foldTallyLocked(slot, ps, base, set)...) - verified = append(verified, p.foldTallyLocked(slot, ps, fb, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, base, set) + p.verifyAndFoldTallyWithLockReleased(slot, ps, fb, set) + if p.slots[slot] != ps { + return nil + } verifiedStake := uint64(0) if base != nil { @@ -866,29 +989,41 @@ func (p *CertPool) maybeAssembleLocked(slot uint64, ps *poolSlot) ([]Certificate continue } p.emitted[key] = struct{}{} - p.snap.CertsEmitted++ emits = append(emits, cert) } - return emits, verified + return emits } -// foldTallyLocked batch-verifies a tally's pending votes. All sign the same +// verifyAndFoldTallyWithLockReleased requires p.mu held on entry and returns +// with p.mu held on every path. The caller must own ps.processing. Verification +// temporarily releases p.mu; retained slot and validator bindings are rechecked +// after reacquiring it before any verified results are installed. +// +// It batch-verifies a tally's pending votes. All sign the same // payload, so randomized weighted pubkey/signature sums need one pairing; // failures bisect to isolate the bad votes (dropped + counted). Only AFTER a // vote verifies does it update the durable per-slot state — the vote-budget / // equivocation ledger (verifiedHash) and base↔fallback disjointness — so raw // votes can never poison those. -func (p *CertPool) foldTallyLocked(slot uint64, ps *poolSlot, tl *tally, set *ValidatorSet) []VerifiedVote { - if tl == nil || len(tl.pending) == 0 { - return nil +func (p *CertPool) verifyAndFoldTallyWithLockReleased(slot uint64, ps *poolSlot, tl *tally, set *ValidatorSet) { + if p.slots[slot] != ps || tl == nil || len(tl.pending) == 0 { + return } batch := make([]VoteMessage, 0, pendingCandidateCount(tl)) + // Reuse admission keys without hashing each signature again on removal. + var keyStorage [64][sha256.Size]byte + keys := keyStorage[:0] + if cap(batch) > len(keyStorage) { + keys = make([][sha256.Size]byte, 0, cap(batch)) + } for rank, candidates := range tl.pending { count := len(candidates) if int(rank) < len(set.Validators) { - for _, msg := range candidates { + for key, msg := range candidates { batch = append(batch, msg) + keys = append(keys, key) } + continue } else { p.snap.VotesRejected += uint64(count) } @@ -900,10 +1035,50 @@ func (p *CertPool) foldTallyLocked(slot uint64, ps *poolSlot, tl *tally, set *Va } p.totalPending -= count } + // Keep candidates in the pending maps while verifying. Concurrent arrivals + // can deduplicate against them, and every in-flight byte still consumes the + // normal per-rank, per-slot and global admission budget. + generation := p.epochGeneration + shredVersion := p.verifier.ShredVersion() + p.mu.Unlock() + p.verificationMu.Lock() good := p.verifyBatch(batch, set) + p.verificationMu.Unlock() + p.mu.Lock() + if p.slots[slot] != ps { + return // pruning/eviction already released the pending accounting + } + current := p.setForSlotLocked(slot) + if generation != p.epochGeneration || shredVersion != p.verifier.ShredVersion() || !sameInstalledValidatorSet(current, set) { + // Do not mix an old aggregate with a new epoch/key/stake binding. Retire + // this slot's old state; a later arrival starts afresh with the current set. + p.totalPending -= ps.pendingCount + delete(p.slots, slot) + for key := range p.emitted { + if key.Slot == slot { + delete(p.emitted, key) + } + } + p.workCond.Broadcast() + return + } + for i, msg := range batch { + candidates := tl.pending[msg.Rank] + delete(candidates, keys[i]) + if len(candidates) == 0 { + delete(tl.pending, msg.Rank) + } + ps.pendingCount-- + ps.pendingByRank[msg.Rank]-- + if ps.pendingByRank[msg.Rank] == 0 { + delete(ps.pendingByRank, msg.Rank) + } + p.totalPending-- + } verified := make([]VerifiedVote, 0, len(good)) - for _, msg := range good { + for i := range good { + msg := good[i].message verified = append(verified, VerifiedVote{ Message: msg, Result: VoteVerifyResult{ @@ -951,40 +1126,47 @@ func (p *CertPool) foldTallyLocked(slot uint64, ps *poolSlot, tl *tally, set *Va } } - pub, err := validatorBLSPubkey(*set, int(msg.Rank)) - if err != nil { - continue - } - var sig bls12381.G2Affine - if _, err := sig.SetBytes(msg.Signature); err != nil { - continue - } - tl.aggPub.Add(&tl.aggPub, &pub) - tl.aggSig.Add(&tl.aggSig, &sig) + // Reuse the exact signature point authenticated by verifyBatch. + tl.aggSig.Add(&tl.aggSig, &good[i].sig) tl.verified[msg.Rank] = struct{}{} tl.stake += set.Validators[msg.Rank].Stake ps.verifiedHash[dk] = append(seen, msg.Vote.BlockHash) } p.snap.BadSignatures += uint64(len(batch) - len(good)) - return verified + // Publish this batch before verifying any newly buffered work. In particular, + // a growing slot must not hold back an already-authenticated quorum. + sink := p.verifiedVoteSink + publication := p.reservePublicationLocked(verified) + p.mu.Unlock() + p.publishVerifiedVotes(sink, verified, publication) + p.mu.Lock() } -func (p *CertPool) foldPendingRankLocked(slot uint64, ps *poolSlot, rank uint16, set *ValidatorSet) []VerifiedVote { - var verified []VerifiedVote +// Installed sets own immutable parsed-key arrays. Reinstallation, even for the +// same epoch and keys, gets a new array and therefore invalidates in-flight work. +func sameInstalledValidatorSet(a, b *ValidatorSet) bool { + return a != nil && b != nil && a.Epoch == b.Epoch && len(a.parsedPubkeys) > 0 && + len(a.parsedPubkeys) == len(b.parsedPubkeys) && &a.parsedPubkeys[0] == &b.parsedPubkeys[0] +} + +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) foldPendingRankLocked(slot uint64, ps *poolSlot, rank uint16, set *ValidatorSet) { for _, tl := range ps.tallies { if len(tl.pending[rank]) != 0 { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) } } - return verified + return } -func (p *CertPool) foldAllPendingLocked(slot uint64, ps *poolSlot, set *ValidatorSet) []VerifiedVote { - var verified []VerifiedVote +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) foldAllPendingLocked(slot uint64, ps *poolSlot, set *ValidatorSet) { for _, tl := range ps.tallies { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) } - return verified + return } func (p *CertPool) evictFartherFutureSlotLocked(incoming uint64) bool { @@ -1017,6 +1199,7 @@ func (p *CertPool) evictFartherFutureSlotLocked(incoming uint64) bool { p.totalPending = 0 } delete(p.slots, victim) + p.workCond.Broadcast() return true } @@ -1045,13 +1228,13 @@ type parsedBatchVote struct { // check, bisecting on failure. Unweighted aggregation is insufficient here: // invalid shares can cancel while the downstream pool later keeps only a subset. // Independent random coefficients bind success to every individual member. -func (p *CertPool) verifyBatch(batch []VoteMessage, set *ValidatorSet) []VoteMessage { +func (p *CertPool) verifyBatch(batch []VoteMessage, set *ValidatorSet) []parsedBatchVote { if len(batch) == 0 { return nil } - p.snap.BatchesVerified++ payload, err := EncodeVotePayloadToSign(batch[0].Vote, p.verifier.ShredVersion()) if err != nil { + p.batchesVerified.Add(1) return nil } @@ -1068,31 +1251,80 @@ func (p *CertPool) verifyBatch(batch []VoteMessage, set *ValidatorSet) []VoteMes members = append(members, parsedBatchVote{message: msg, pubkey: pub, sig: sig}) } + return p.verifyParsedBatch(members, payload) +} + +// verifyParsedBatch keeps the owned parsed points through failed-batch +// subdivision and returns only individually bound, verified members. Each +// subdivision still uses fresh random coefficients; parsing is the only work +// reused. Verification does not read or mutate the pool's slot maps. +func (p *CertPool) verifyParsedBatch(members []parsedBatchVote, payload []byte) []parsedBatchVote { + p.batchesVerified.Add(1) if len(members) == 0 { return nil } if len(members) == 1 { if aggregatePairingOK(members[0].pubkey, payload, members[0].sig) { - return []VoteMessage{members[0].message} + return members } return nil } if ok, err := randomizedAggregatePairingOK(members, payload); err == nil && ok { - return batchMessages(members) + return members } else if err != nil { // Entropy failure must reduce performance, never verification strength. return individuallyVerifiedBatch(members, payload) } // Aggregate failed: bisect the structurally-valid subset. - messages := batchMessages(members) - mid := len(messages) / 2 - valid := p.verifyBatch(messages[:mid], set) - valid = append(valid, p.verifyBatch(messages[mid:], set)...) + mid := len(members) / 2 + valid := p.verifyParsedBatch(members[:mid:mid], payload) + valid = append(valid, p.verifyParsedBatch(members[mid:], payload)...) return valid } func randomizedAggregatePairingOK(members []parsedBatchVote, payload []byte) (bool, error) { + // Bucket setup outweighs MultiExp's savings on small batches, including + // the two-candidate collision path and failed-batch subdivisions. Its + // internal task handoffs can also delay verification behind replay when + // only one Go execution thread is available, despite doing less arithmetic. + if len(members) < 16 || runtime.GOMAXPROCS(0) == 1 { + return randomizedAggregatePairingScalarOK(members, payload) + } + return randomizedAggregatePairingMultiExpOK(members, payload) +} + +func randomizedAggregatePairingMultiExpOK(members []parsedBatchVote, payload []byte) (bool, error) { + pubkeys := make([]bls12381.G1Affine, len(members)) + signatures := make([]bls12381.G2Affine, len(members)) + coefficients := make([]blsfr.Element, len(members)) + for i := range members { + coefficient, err := randomNonzeroBatchCoefficient() + if err != nil { + return false, err + } + // Keep independent, nonzero, full-field coefficients and apply the + // same coefficient to each member's public key and signature. + coefficients[i].SetBigInt(coefficient) + pubkeys[i] = members[i].pubkey + signatures[i] = members[i].sig + } + // MultiExp defaults to using all CPUs. Keep its arithmetic concurrency at + // one, and run G1/G2 sequentially, to avoid competing with replay workers. + // Pool-level verification admission remains bounded by verificationMu. + config := ecc.MultiExpConfig{NbTasks: 1} + var aggPub bls12381.G1Affine + if _, err := aggPub.MultiExp(pubkeys, coefficients, config); err != nil { + return false, err + } + var aggSig bls12381.G2Affine + if _, err := aggSig.MultiExp(signatures, coefficients, config); err != nil { + return false, err + } + return aggregatePairingOK(aggPub, payload, aggSig), nil +} + +func randomizedAggregatePairingScalarOK(members []parsedBatchVote, payload []byte) (bool, error) { var aggPub bls12381.G1Affine var aggSig bls12381.G2Affine aggPub.SetInfinity() @@ -1124,24 +1356,16 @@ func randomNonzeroBatchCoefficient() (*big.Int, error) { } } -func individuallyVerifiedBatch(members []parsedBatchVote, payload []byte) []VoteMessage { - valid := make([]VoteMessage, 0, len(members)) +func individuallyVerifiedBatch(members []parsedBatchVote, payload []byte) []parsedBatchVote { + valid := make([]parsedBatchVote, 0, len(members)) for i := range members { if aggregatePairingOK(members[i].pubkey, payload, members[i].sig) { - valid = append(valid, members[i].message) + valid = append(valid, members[i]) } } return valid } -func batchMessages(members []parsedBatchVote) []VoteMessage { - messages := make([]VoteMessage, len(members)) - for i := range members { - messages[i] = members[i].message - } - return messages -} - func aggregatePairingOK(aggPub bls12381.G1Affine, payload []byte, aggSig bls12381.G2Affine) bool { if aggPub.IsInfinity() { return false diff --git a/pkg/alpenglow/certpool_bench_test.go b/pkg/alpenglow/certpool_bench_test.go new file mode 100644 index 000000000..a6864f0af --- /dev/null +++ b/pkg/alpenglow/certpool_bench_test.go @@ -0,0 +1,160 @@ +package alpenglow + +import ( + "crypto/sha256" + "fmt" + "slices" + "sync" + "sync/atomic" + "testing" + "time" + + bls12381 "github.com/Overclock-Validator/gnark-crypto/ecc/bls12-381" + "github.com/gagliardetto/solana-go" +) + +// Benchmark the full pending-vote fold, including verification, aggregation and +// accounting. Keys and signatures are prepared outside the timed region. +func BenchmarkCertPoolFoldVerifiedBatch(b *testing.B) { + for _, size := range []int{1, 8, 32, 64} { + b.Run(fmt.Sprintf("votes=%d", size), func(b *testing.B) { + verifier, installed, vote, batch := certPoolBenchmarkFixture(b, size) + b.ReportAllocs() + b.ResetTimer() + for n := 0; n < b.N; n++ { + pool := NewCertPool(DefaultCertPoolConfig(), verifier, nil) + pool.SetEpochLookup(func(uint64) uint64 { return installed.Epoch }) + tl := newTally() + ps := &poolSlot{processing: true, verifiedHash: make(map[voteDedupKey][]solana.Hash), pendingByRank: make(map[uint16]int)} + pool.slots[vote.Slot] = ps + for _, msg := range batch { + tl.pending[msg.Rank] = map[[sha256.Size]byte]VoteMessage{sha256.Sum256(msg.Signature): msg} + ps.pendingByRank[msg.Rank]++ + ps.pendingCount++ + pool.totalPending++ + } + pool.mu.Lock() + pool.verifyAndFoldTallyWithLockReleased(vote.Slot, ps, tl, &installed) + pool.mu.Unlock() + if len(tl.verified) != size || tl.stake != uint64(size) || pool.totalPending != 0 { + b.Fatal("incomplete verified fold") + } + } + }) + } +} + +func certPoolBenchmarkFixture(b testing.TB, size int) (*CertificateVerifier, ValidatorSet, Vote, []VoteMessage) { + stakes := make([]uint64, size) + for i := range stakes { + stakes[i] = 1 + } + set, keys := testBLSValidatorSet(uint64(size), stakes...) + verifier := NewCertificateVerifier() + if err := verifier.SetValidatorSet(set); err != nil { + b.Fatal(err) + } + // Exercise the same cached public-key representation used in production. + installed, ok := verifier.ValidatorSetForEpoch(set.Epoch) + if !ok { + b.Fatal("missing installed validator set") + } + vote := NewSkipVote(500) + payload, err := EncodeVotePayloadToSign(vote, verifier.ShredVersion()) + if err != nil { + b.Fatal(err) + } + point, err := bls12381.HashToG2(payload, []byte(blsHashToPointDST)) + if err != nil { + b.Fatal(err) + } + batch := make([]VoteMessage, size) + for i := range batch { + var sig bls12381.G2Affine + sig.ScalarMultiplication(&point, keys[i]) + raw := sig.RawBytes() + batch[i] = VoteMessage{Vote: vote, Rank: uint16(i), Signature: raw[:]} + } + return verifier, installed, vote, batch +} + +// Observe short pool reads while eight peer-like producers submit one slot. +// The latency metrics, not ns/op (which includes deliberate sampling delays), +// measure whether cryptographic work blocks unrelated pool readers. +func BenchmarkCertPoolConcurrentVotes(b *testing.B) { + verifier, installed, vote, batch := certPoolBenchmarkFixture(b, 64) + waits := make([]int64, 0, 16*b.N) + b.ResetTimer() + for n := 0; n < b.N; n++ { + pool := NewCertPool(DefaultCertPoolConfig(), verifier, nil) + pool.SetEpochLookup(func(uint64) uint64 { return installed.Epoch }) + pool.NoteLiveSlot(vote.Slot) + var verified atomic.Int64 + pool.SetVerifiedVoteSink(func(VerifiedVote) { verified.Add(1) }) + start := make(chan struct{}) + var producers sync.WaitGroup + for worker := 0; worker < 8; worker++ { + producers.Add(1) + go func(worker int) { + defer producers.Done() + <-start + for i := worker; i < len(batch); i += 8 { + pool.AddVote(batch[i]) + } + }(worker) + } + close(start) + for sample := 0; sample < 16; sample++ { + time.Sleep(100 * time.Microsecond) + t := time.Now() + pool.Snapshot() + waits = append(waits, time.Since(t).Nanoseconds()) + } + producers.Wait() + pool.FlushRewardVotes(vote.Slot) + if verified.Load() != 64 || pool.Snapshot().PendingTotal != 0 { + b.Fatal("lost or duplicated votes") + } + } + b.StopTimer() + slices.Sort(waits) + b.ReportMetric(float64(waits[len(waits)*95/100]), "snapshot-p95-ns") + b.ReportMetric(float64(waits[len(waits)-1]), "snapshot-max-ns") +} + +func BenchmarkCertPoolWeightedPairing(b *testing.B) { + for _, size := range []int{2, 4, 8, 16, 32, 64, 128} { + verifier, set, vote, batch := certPoolBenchmarkFixture(b, size) + payload, err := EncodeVotePayloadToSign(vote, verifier.ShredVersion()) + if err != nil { + b.Fatal(err) + } + members := make([]parsedBatchVote, size) + for i, msg := range batch { + pub, err := validatorBLSPubkey(set, int(msg.Rank)) + if err != nil { + b.Fatal(err) + } + members[i] = parsedBatchVote{message: msg, pubkey: pub} + if _, err := members[i].sig.SetBytes(msg.Signature); err != nil { + b.Fatal(err) + } + } + for _, impl := range []struct { + name string + verify func([]parsedBatchVote, []byte) (bool, error) + }{ + {"Scalar", randomizedAggregatePairingScalarOK}, + {"MultiExp", randomizedAggregatePairingMultiExpOK}, + } { + b.Run(fmt.Sprintf("votes=%d/%s", size, impl.name), func(b *testing.B) { + b.ReportAllocs() + for n := 0; n < b.N; n++ { + if ok, err := impl.verify(members, payload); err != nil || !ok { + b.Fatalf("verification: %t %v", ok, err) + } + } + }) + } + } +} diff --git a/pkg/alpenglow/certpool_concurrency_test.go b/pkg/alpenglow/certpool_concurrency_test.go new file mode 100644 index 000000000..a97c9eea9 --- /dev/null +++ b/pkg/alpenglow/certpool_concurrency_test.go @@ -0,0 +1,252 @@ +package alpenglow + +import ( + "fmt" + "sync" + "testing" + "time" +) + +func waitCertPoolCall(t *testing.T, done <-chan struct{}) { + t.Helper() + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("certificate-pool operation did not finish") + } +} + +func certPoolAsync(fn func()) <-chan struct{} { + done := make(chan struct{}) + go func() { defer close(done); fn() }() + return done +} + +// Park a real owner at the production verification gate. No cryptography is +// stubbed; releasing the gate runs the normal randomized verification path. +func parkCertPoolVerification(t *testing.T, pool *CertPool, msg VoteMessage) (func(), <-chan struct{}) { + t.Helper() + pool.verificationMu.Lock() + var once sync.Once + release := func() { once.Do(pool.verificationMu.Unlock) } + t.Cleanup(release) + done := certPoolAsync(func() { pool.AddVote(msg) }) + ready := certPoolAsync(func() { + for pool.Snapshot().PendingTotal == 0 { + time.Sleep(time.Millisecond) + } + }) + waitCertPoolCall(t, ready) // Also proves Snapshot can acquire mu during work. + return release, done +} + +func TestCertPoolOffLockAdmissionAndRewardFlush(t *testing.T) { + pool, set, keys, emitted := newTestPool(t) + seen := make(chan VerifiedVote, 8) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { seen <- v }) + vote := NewSkipVote(500) + first := VoteMessage{Vote: vote, Rank: 0, Signature: signTestVote(t, vote, keys[0])} + second := VoteMessage{Vote: vote, Rank: 1, Signature: signTestVote(t, vote, keys[1])} + release, owner := parkCertPoolVerification(t, pool, first) + waitCertPoolCall(t, certPoolAsync(func() { pool.AddVote(second) })) + if got := pool.Snapshot().PendingTotal; got != 2 { + t.Fatalf("pending = %d; in-flight and newly buffered votes must both count", got) + } + flush := certPoolAsync(func() { pool.FlushRewardVotes(vote.Slot) }) + select { + case <-flush: + t.Fatal("flush escaped in-flight verification") + case <-time.After(20 * time.Millisecond): + } + if len(seen) != 0 { + t.Fatal("unverified vote reached sink") + } + release() + waitCertPoolCall(t, owner) + waitCertPoolCall(t, flush) + if len(seen) != 2 || pool.Snapshot().PendingTotal != 0 { + t.Fatalf("flush lost or duplicated work: published=%d snapshot=%+v", len(seen), pool.Snapshot()) + } + if len(*emitted) == 0 { + t.Fatal("buffered arrivals did not drive certificate assembly") + } + for _, cert := range *emitted { + if _, _, err := verifyCertificateWithSet(set, cert, true); err != nil { + t.Fatalf("concurrently assembled certificate failed verification: %v", err) + } + } +} + +func TestCertPoolOffLockPruningDiscardsInFlightResults(t *testing.T) { + pool, _, keys, emitted := newTestPool(t) + seen := make(chan VerifiedVote, 8) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { seen <- v }) + vote := NewSkipVote(500) + release, owner := parkCertPoolVerification(t, pool, VoteMessage{Vote: vote, Rank: 0, Signature: signTestVote(t, vote, keys[0])}) + waitCertPoolCall(t, certPoolAsync(func() { pool.ObserveFloor(vote.Slot) })) + if got := pool.Snapshot(); got.PendingTotal != 0 || got.Slots != 0 { + t.Fatalf("pruning retained in-flight accounting: %+v", got) + } + release() + waitCertPoolCall(t, owner) + if len(seen) != 0 || len(*emitted) != 0 || pool.Snapshot().PendingTotal != 0 { + t.Fatal("pruned work was published or decremented accounting twice") + } + addVote(t, pool, NewSkipVote(501), 0, keys[0]) + if len(seen) != 1 { + t.Fatal("subsequent live vote was lost") + } +} + +func TestCertPoolOffLockBindingChangeDiscardsResults(t *testing.T) { + for _, kind := range []string{"validator-set", "epoch-lookup"} { + t.Run(kind, func(t *testing.T) { + pool, set, keys, emitted := newTestPool(t) + seen := make(chan VerifiedVote, 8) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { seen <- v }) + vote := NewSkipVote(500) + msg := VoteMessage{Vote: vote, Rank: 0, Signature: signTestVote(t, vote, keys[0])} + release, owner := parkCertPoolVerification(t, pool, msg) + if kind == "validator-set" { + if err := pool.verifier.SetValidatorSet(set); err != nil { + t.Fatal(err) + } + } else { + pool.SetEpochLookup(func(uint64) uint64 { return set.Epoch }) + } + release() + waitCertPoolCall(t, owner) + if got := pool.Snapshot(); len(seen) != 0 || len(*emitted) != 0 || got.PendingTotal != 0 || got.Slots != 0 { + t.Fatalf("stale binding published or retained state: %+v", got) + } + pool.AddVote(msg) + if len(seen) != 1 { + t.Fatal("vote could not be retried under the current binding") + } + }) + } +} + +func TestCertPoolOffLockPendingBoundAndDuplicates(t *testing.T) { + pool, _, keys, _ := newTestPool(t) + pool.cfg.MaxPendingVotesPerSlot = 2 + pool.cfg.MaxPendingVotesTotal = 2 + vote := NewSkipVote(500) + msgs := make([]VoteMessage, 3) + for i := range msgs { + msgs[i] = VoteMessage{Vote: vote, Rank: uint16(i), Signature: signTestVote(t, vote, keys[i])} + } + release, owner := parkCertPoolVerification(t, pool, msgs[0]) + waitCertPoolCall(t, certPoolAsync(func() { pool.AddVote(msgs[1]) })) + waitCertPoolCall(t, certPoolAsync(func() { pool.AddVote(msgs[0]) })) + third := certPoolAsync(func() { pool.AddVote(msgs[2]) }) + select { + case <-third: + t.Fatal("capacity-pressure admission did not wait for authentication") + case <-time.After(20 * time.Millisecond): + } + if got := pool.Snapshot().PendingTotal; got != 2 { + t.Fatalf("in-flight votes escaped bounds or duplicate consumed capacity: %d", got) + } + release() + waitCertPoolCall(t, owner) + waitCertPoolCall(t, third) + pool.FlushRewardVotes(vote.Slot) + if got := pool.Snapshot(); got.PendingTotal != 0 || got.VotesAccepted != 3 { + t.Fatalf("capacity wakeup lost or duplicated votes: %+v", got) + } +} + +func TestCertPoolOffLockEvictionDoesNotResurrectSlot(t *testing.T) { + set, keys := testBLSValidatorSet(100, 40, 30, 15, 10, 5) + verifier := NewCertificateVerifier() + if err := verifier.SetValidatorSet(set); err != nil { + t.Fatal(err) + } + pool := NewCertPool(CertPoolConfig{MaxLiveSlots: 1}, verifier, nil) + pool.SetEpochLookup(func(uint64) uint64 { return set.Epoch }) + pool.NoteLiveSlot(100) + seen := make(chan VerifiedVote, 8) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { seen <- v }) + future := NewSkipVote(110) + release, owner := parkCertPoolVerification(t, pool, VoteMessage{Vote: future, Rank: 0, Signature: signTestVote(t, future, keys[0])}) + near := NewSkipVote(101) + msg := VoteMessage{Vote: near, Rank: 4, Signature: signTestVote(t, near, keys[4])} + waitCertPoolCall(t, certPoolAsync(func() { pool.AddVote(msg) })) + if got := pool.Snapshot(); got.PendingTotal != 1 || got.Slots != 1 { + t.Fatalf("eviction accounting mismatch: %+v", got) + } + release() + waitCertPoolCall(t, owner) + pool.FlushRewardVotes(near.Slot) + if len(seen) != 1 || (<-seen).Message.Vote.Slot != near.Slot || pool.Snapshot().PendingTotal != 0 { + t.Fatal("evicted work displaced or corrupted the nearer slot") + } +} + +func TestCertPoolRewardFlushPriority(t *testing.T) { + for _, prune := range []bool{false, true} { + t.Run(fmt.Sprintf("prune=%t", prune), func(t *testing.T) { + pool, _, keys, _ := newTestPool(t) + vote := NewSkipVote(500) + // A below-threshold vote must be published by the reward flush. + pool.AddVote(VoteMessage{Vote: vote, Rank: 4, Signature: signTestVote(t, vote, keys[4])}) + pool.mu.Lock() + ps := pool.slots[vote.Slot] + ps.processing = true // Hold ownership until both flushes are queued. + pool.mu.Unlock() + flush1 := certPoolAsync(func() { pool.FlushRewardVotes(vote.Slot) }) + flush2 := certPoolAsync(func() { pool.FlushRewardVotes(vote.Slot) }) + deadline := time.Now().Add(2 * time.Second) + for { + pool.mu.Lock() + waiting := ps.flushWaiting + pool.mu.Unlock() + if waiting == 2 { + break + } + if time.Now().After(deadline) { + t.Fatal("flushes did not register") + } + time.Sleep(time.Millisecond) + } + entered, release := make(chan struct{}), make(chan struct{}) + var once sync.Once + t.Cleanup(func() { once.Do(func() { close(release) }) }) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { + if v.Message.Rank == 4 { + close(entered) + <-release + } + }) + msg := VoteMessage{Vote: vote, Rank: 0, Signature: signTestVote(t, vote, keys[0])} + arrival := certPoolAsync(func() { pool.AddVote(msg) }) + if prune { + pool.ObserveFloor(vote.Slot) + } else { + pool.mu.Lock() + ps.processing = false + pool.workCond.Broadcast() + pool.mu.Unlock() + waitCertPoolCall(t, entered) + if got := pool.Snapshot().VotesAccepted; got != 1 { + t.Fatalf("new arrival overtook reward flush: accepted=%d", got) + } + select { + case <-arrival: + t.Fatal("arrival escaped active flush") + default: + } + } + once.Do(func() { close(release) }) + waitCertPoolCall(t, flush1) + waitCertPoolCall(t, flush2) + waitCertPoolCall(t, arrival) + pool.mu.Lock() + defer pool.mu.Unlock() + if ps.flushWaiting != 0 { + t.Fatalf("leaked flush waiters: %d", ps.flushWaiting) + } + }) + } +} diff --git a/pkg/alpenglow/certpool_points_test.go b/pkg/alpenglow/certpool_points_test.go new file mode 100644 index 000000000..4f76ad01e --- /dev/null +++ b/pkg/alpenglow/certpool_points_test.go @@ -0,0 +1,102 @@ +package alpenglow + +import ( + "bytes" + "testing" + + bls12381 "github.com/Overclock-Validator/gnark-crypto/ecc/bls12-381" +) + +func TestCertPoolVerifiedPointsMatchIndividualVerification(t *testing.T) { + for _, mixed := range []bool{false, true} { + name := "valid" + if mixed { + name = "mixed-invalid" + } + t.Run(name, func(t *testing.T) { + pool, set, keys, _ := newTestPool(t) + vote := NewSkipVote(500) + var batch []VoteMessage + for i, key := range keys { + signedVote := vote + if mixed && i%2 == 1 { + signedVote = NewSkipVote(501) // Valid point, wrong payload, in both halves. + } + batch = append(batch, VoteMessage{Vote: vote, Rank: uint16(i), Signature: signTestVote(t, signedVote, key)}) + } + if mixed { + var infinity bls12381.G2Affine + infinity.SetInfinity() + raw := infinity.RawBytes() + batch = append(batch, + VoteMessage{Vote: vote, Rank: 0, Signature: []byte{0xff}}, + VoteMessage{Vote: vote, Rank: 1, Signature: raw[:]}, + VoteMessage{Vote: vote, Rank: uint16(len(keys)), Signature: batch[0].Signature}, + ) + } + var expected []VoteMessage + for _, msg := range batch { + if _, err := verifyVoteMessageWithSet(set, msg); err == nil { + expected = append(expected, msg) + } + } + verified := pool.verifyBatch(batch, &set) + if len(verified) != len(expected) { + t.Fatalf("verified %d members, want %d", len(verified), len(expected)) + } + for i, member := range verified { + if member.message.Rank != expected[i].Rank || member.message.Vote != expected[i].Vote || !bytes.Equal(member.message.Signature, expected[i].Signature) { + t.Fatalf("member %d does not match its individually verified message", i) + } + raw := member.sig.RawBytes() + if !bytes.Equal(raw[:], expected[i].Signature) { + t.Fatalf("member %d retained the wrong signature point", i) + } + pub, err := validatorBLSPubkey(set, int(expected[i].Rank)) + if err != nil || !member.pubkey.Equal(&pub) { + t.Fatalf("member %d retained the wrong public key", i) + } + } + if mixed && pool.Snapshot().BatchesVerified <= 1 { + t.Fatal("mixed batch did not exercise recursive verification") + } + }) + } +} + +// Entropy failure uses this individual-verification path instead of accepting +// an unweighted aggregate. Reusing points must preserve its filtering too. +func TestIndividuallyVerifiedBatchRetainsOnlyValidPoints(t *testing.T) { + _, set, keys, _ := newTestPool(t) + vote := NewSkipVote(502) + payload, err := EncodeVotePayloadToSign(vote, 0) + if err != nil { + t.Fatal(err) + } + var members []parsedBatchVote + for i, key := range keys { + signedVote := vote + if i%2 == 1 { + signedVote = NewSkipVote(503) + } + msg := VoteMessage{Vote: vote, Rank: uint16(i), Signature: signTestVote(t, signedVote, key)} + pub, err := validatorBLSPubkey(set, i) + if err != nil { + t.Fatal(err) + } + var sig bls12381.G2Affine + if _, err := sig.SetBytes(msg.Signature); err != nil { + t.Fatal(err) + } + members = append(members, parsedBatchVote{message: msg, pubkey: pub, sig: sig}) + } + verified := individuallyVerifiedBatch(members, payload) + if len(verified) != 3 { + t.Fatalf("verified %d, want 3", len(verified)) + } + for i, member := range verified { + if member.message.Rank != uint16(i*2) || !member.sig.Equal(&members[i*2].sig) || !member.pubkey.Equal(&members[i*2].pubkey) { + t.Fatalf("wrong verified member at %d", i) + } + } +} diff --git a/pkg/alpenglow/certpool_progress_test.go b/pkg/alpenglow/certpool_progress_test.go new file mode 100644 index 000000000..6732dce4b --- /dev/null +++ b/pkg/alpenglow/certpool_progress_test.go @@ -0,0 +1,112 @@ +package alpenglow + +import ( + "sync" + "testing" + "time" +) + +func TestCertPoolProgressDoesNotWaitForVerificationLock(t *testing.T) { + pool := NewCertPool(DefaultCertPoolConfig(), NewCertificateVerifier(), nil) + pool.mu.Lock() // The same lock held while verifying incoming BLS votes. + done := make(chan struct{}) + go func() { + pool.NoteLiveSlot(123) + close(done) + }() + completed := false + select { + case <-done: + completed = true + case <-time.After(time.Second): + } + pool.mu.Unlock() + <-done + if !completed { + t.Fatal("trusted replay progress waited for the verification lock") + } + if got := pool.Snapshot().LiveSlot; got != 123 { + t.Fatalf("live slot = %d, want 123", got) + } +} + +func TestCertPoolConcurrentProgressRemainsMonotonic(t *testing.T) { + pool := NewCertPool(DefaultCertPoolConfig(), NewCertificateVerifier(), nil) + const workers, updates = 16, 256 + start := make(chan struct{}) + var wg sync.WaitGroup + for worker := range workers { + wg.Go(func() { + <-start + for n := range updates { + slot := uint64(n*workers + worker + 1) + pool.NoteLiveSlot(slot) + pool.NoteLiveSlot(slot / 2) // Stale updates race newer progress. + } + }) + } + finished := make(chan struct{}) + go func() { + wg.Wait() + close(finished) + }() + close(start) + var previous uint64 + for { + live := pool.Snapshot().LiveSlot + pool.mu.Lock() + anchor := pool.windowAnchorLocked() + pool.mu.Unlock() + if live < previous || anchor < live { + t.Errorf("progress regressed: previous=%d snapshot=%d anchor=%d", previous, live, anchor) + } + previous = anchor + select { + case <-finished: + pool.NoteLiveSlot(0) + pool.NoteLiveSlot(1) + if got := pool.Snapshot().LiveSlot; got != workers*updates { + t.Fatalf("live slot = %d, want %d", got, workers*updates) + } + return + default: + } + } +} + +func TestCertPoolProgressPreservesTrustedVoteWindow(t *testing.T) { + set, keys := testBLSValidatorSet(100, 40, 30, 15, 10, 5) + verifier := NewCertificateVerifier() + if err := verifier.SetValidatorSet(set); err != nil { + t.Fatal(err) + } + pool := NewCertPool(CertPoolConfig{MaxSlotsAhead: 10}, verifier, nil) + pool.SetEpochLookup(func(uint64) uint64 { return set.Epoch }) + pool.NoteLiveSlot(100) + checkVote := func(slot uint64, rejected bool, live uint64) { + t.Helper() + before := pool.Snapshot() + addVote(t, pool, NewSkipVote(slot), 4, keys[4]) + after := pool.Snapshot() + wantRejected := before.VotesRejected + if rejected { + wantRejected++ + } + if after.VotesRejected != wantRejected { + t.Fatalf("slot %d: rejected=%d, want %d", slot, after.VotesRejected, wantRejected) + } + if after.LiveSlot != live { + t.Fatalf("raw vote at %d moved trusted progress to %d, want %d", slot, after.LiveSlot, live) + } + } + checkVote(110, false, 100) // Inclusive upper edge. + checkVote(111, true, 100) // Accepted raw votes cannot slide the window. + pool.NoteLiveSlot(90) + checkVote(111, true, 100) + pool.NoteLiveSlot(101) + checkVote(111, false, 101) + pool.ObserveFloor(120) + checkVote(120, true, 101) // Finalized floor still rejects old votes. + checkVote(130, false, 101) // Floor can anchor the window above replay. + checkVote(131, true, 101) +} diff --git a/pkg/alpenglow/certpool_test.go b/pkg/alpenglow/certpool_test.go index 8c7c36789..24970345c 100644 --- a/pkg/alpenglow/certpool_test.go +++ b/pkg/alpenglow/certpool_test.go @@ -1,6 +1,7 @@ package alpenglow import ( + "fmt" "math/big" "testing" "time" @@ -805,3 +806,50 @@ func TestCertPoolEmitsOnce(t *testing.T) { t.Fatalf("no new certs expected, went from %d to %d", n, len(*emitted)) } } + +// Exercise both aggregation paths, including a malformed member at every +// position, a different signed payload, and failed-batch subdivision. Valid +// votes must survive independently of which member causes the batch to fail. +func TestCertPoolBatchAggregationPaths(t *testing.T) { + for _, size := range []int{2, 8, 16, 64} { + t.Run(fmt.Sprintf("votes=%d", size), func(t *testing.T) { + verifier, set, vote, batch := certPoolBenchmarkFixture(t, size) + pool := NewCertPool(DefaultCertPoolConfig(), verifier, nil) + payload, err := EncodeVotePayloadToSign(vote, verifier.ShredVersion()) + if err != nil { + t.Fatal(err) + } + members := pool.verifyBatch(batch, &set) + if len(members) != size { + t.Fatal("valid batch rejected") + } + var tweak bls12381.G2Affine + tweak.ScalarMultiplicationBase(big.NewInt(1234567)) + for i := range members { + bad := append([]parsedBatchVote(nil), members...) + bad[i].sig.Add(&bad[i].sig, &tweak) + if ok, err := randomizedAggregatePairingOK(bad, payload); err != nil || ok { + t.Fatalf("bad member %d: accepted=%t err=%v", i, ok, err) + } + } + wrongPayload := append([]byte(nil), payload...) + wrongPayload[0] ^= 1 + if ok, err := randomizedAggregatePairingOK(members, wrongPayload); err != nil || ok { + t.Fatalf("wrong payload: accepted=%t err=%v", ok, err) + } + // Preserve the unweighted sum while corrupting two shares in a + // large batch; subdivision must keep exactly the honest members. + members[0].sig.Add(&members[0].sig, &tweak) + members[len(members)-1].sig.Sub(&members[len(members)-1].sig, &tweak) + valid := pool.verifyParsedBatch(members, payload) + if len(valid) != size-2 { + t.Fatalf("verified %d, want %d", len(valid), size-2) + } + for _, member := range valid { + if member.message.Rank == 0 || int(member.message.Rank) == size-1 { + t.Fatal("invalid share survived") + } + } + }) + } +} diff --git a/pkg/alpenglow/observer.go b/pkg/alpenglow/observer.go index 9b84e58cb..8f3470bec 100644 --- a/pkg/alpenglow/observer.go +++ b/pkg/alpenglow/observer.go @@ -95,6 +95,17 @@ type Observer struct { replayBlocks map[uint64]BlockID replayOrder []uint64 replayChecks map[CertificateKey]certificateReplayCheck + // Only retained, block-bearing certificates that have not yet been checked + // against replay belong here. Checked history stays in certificates and + // replayChecks for diagnostics/deduplication, but need not be scanned on + // every block (including skipped slots). This is an in-memory observer + // index, not voting authorization or durable crash-recovery state. + pendingReplayCertificates map[CertificateKey]BlockID + // Votes do not change certificate/replay reconciliation. Reuse its exact + // statistics until one of those inputs changes instead of scanning every + // retained certificate for each incoming vote. + pendingStats certificateReplayPendingStats + pendingStatsValid bool votesObserved uint64 certificatesObserved uint64 @@ -149,11 +160,12 @@ func NewObserverWithConfig(cfg ObserverConfig) *Observer { cfg.MaxTrackedReplayBlocks = DefaultMaxTrackedReplayBlocks } return &Observer{ - cfg: cfg, - votes: make(map[VoteMessageKey]VoteMessage), - certificates: make(map[CertificateKey]Certificate), - replayBlocks: make(map[uint64]BlockID), - replayChecks: make(map[CertificateKey]certificateReplayCheck), + cfg: cfg, + votes: make(map[VoteMessageKey]VoteMessage), + certificates: make(map[CertificateKey]Certificate), + replayBlocks: make(map[uint64]BlockID), + replayChecks: make(map[CertificateKey]certificateReplayCheck), + pendingReplayCertificates: make(map[CertificateKey]BlockID), } } @@ -211,6 +223,7 @@ func (o *Observer) ObserveCertificate(cert Certificate) (Observation, error) { key := cert.Key() _, exists := o.certificates[key] if !exists { + o.pendingStatsValid = false tracked := o.trackCertificateLocked(key, cert) o.certificatesObserved++ o.applyCertificateLocked(cert) @@ -225,6 +238,7 @@ func (o *Observer) ObserveCertificate(cert Certificate) (Observation, error) { func (o *Observer) ObserveReplayBlock(obs ReplayBlockObservation) Observation { o.mu.Lock() defer o.mu.Unlock() + o.pendingStatsValid = false if obs.At.IsZero() { obs.At = time.Now() @@ -260,8 +274,8 @@ func (o *Observer) ObserveReplayResult(obs ReplayResultObservation) Observation } func (o *Observer) Snapshot() Snapshot { - o.mu.RLock() - defer o.mu.RUnlock() + o.mu.Lock() + defer o.mu.Unlock() return o.snapshotLocked() } @@ -339,12 +353,16 @@ func (o *Observer) trackCertificateLocked(key CertificateKey, cert Certificate) return false } o.certificates[key] = cert + if block, ok := cert.Block(); ok && block.HasHash() { + o.pendingReplayCertificates[key] = block + } o.certOrder = append(o.certOrder, key) for len(o.certificates) > o.cfg.MaxTrackedCertificates { old := o.certOrder[0] o.certOrder = o.certOrder[1:] delete(o.certificates, old) delete(o.replayChecks, old) + delete(o.pendingReplayCertificates, old) } return true } @@ -369,12 +387,10 @@ func (o *Observer) checkReplayBlockCertificatesLocked(block BlockID) { if !block.HasHash() { return } - for key, cert := range o.certificates { - certBlock, ok := cert.Block() - if !ok || certBlock.Slot != block.Slot { - continue + for key, certBlock := range o.pendingReplayCertificates { + if certBlock.Slot == block.Slot { + o.checkCertificateReplayLocked(key, o.certificates[key]) } - o.checkCertificateReplayLocked(key, cert) } } @@ -410,6 +426,7 @@ func (o *Observer) checkCertificateReplayLocked(key CertificateKey, cert Certifi } } o.replayChecks[key] = check + delete(o.pendingReplayCertificates, key) } type certificateReplayPendingStats struct { @@ -423,30 +440,24 @@ type certificateReplayPendingStats struct { func (o *Observer) certificateReplayPendingStatsLocked() certificateReplayPendingStats { var stats certificateReplayPendingStats - for key, cert := range o.certificates { - if _, checked := o.replayChecks[key]; checked { - continue + for _, certBlock := range o.pendingReplayCertificates { + stats.count++ + if stats.oldestSlot == 0 || certBlock.Slot < stats.oldestSlot { + stats.oldestSlot = certBlock.Slot } - certBlock, ok := cert.Block() - if ok && certBlock.HasHash() { - stats.count++ - if stats.oldestSlot == 0 || certBlock.Slot < stats.oldestSlot { - stats.oldestSlot = certBlock.Slot - } - if certBlock.Slot > stats.newestSlot { - stats.newestSlot = certBlock.Slot - } - if o.oldestReplayBlockSlot != 0 && certBlock.Slot < o.oldestReplayBlockSlot { - stats.preWindow++ - } - if o.oldestReplayBlockSlot != 0 && - o.latestReplayBlockSlot != 0 && - certBlock.Slot >= o.oldestReplayBlockSlot && - certBlock.Slot <= o.latestReplayBlockSlot { - stats.mature++ - if stats.matureOldestSlot == 0 || certBlock.Slot < stats.matureOldestSlot { - stats.matureOldestSlot = certBlock.Slot - } + if certBlock.Slot > stats.newestSlot { + stats.newestSlot = certBlock.Slot + } + if o.oldestReplayBlockSlot != 0 && certBlock.Slot < o.oldestReplayBlockSlot { + stats.preWindow++ + } + if o.oldestReplayBlockSlot != 0 && + o.latestReplayBlockSlot != 0 && + certBlock.Slot >= o.oldestReplayBlockSlot && + certBlock.Slot <= o.latestReplayBlockSlot { + stats.mature++ + if stats.matureOldestSlot == 0 || certBlock.Slot < stats.matureOldestSlot { + stats.matureOldestSlot = certBlock.Slot } } } @@ -454,7 +465,11 @@ func (o *Observer) certificateReplayPendingStatsLocked() certificateReplayPendin } func (o *Observer) snapshotLocked() Snapshot { - pending := o.certificateReplayPendingStatsLocked() + if !o.pendingStatsValid { + o.pendingStats = o.certificateReplayPendingStatsLocked() + o.pendingStatsValid = true + } + pending := o.pendingStats return Snapshot{ VotesObserved: o.votesObserved, CertificatesObserved: o.certificatesObserved, diff --git a/pkg/alpenglow/observer_bench_test.go b/pkg/alpenglow/observer_bench_test.go new file mode 100644 index 000000000..a792015cd --- /dev/null +++ b/pkg/alpenglow/observer_bench_test.go @@ -0,0 +1,57 @@ +package alpenglow + +import ( + "fmt" + "testing" +) + +func BenchmarkObserverVoteWithRetainedCertificates(b *testing.B) { + o := NewObserver() + for slot := uint64(1); slot <= DefaultMaxTrackedCertificates; slot++ { + _, err := o.ObserveCertificate(Certificate{Type: CertificateNotarize, Slot: slot, + BlockHash: testHash(1), IncludedStake: 80, TotalStake: 100}) + if err != nil { + b.Fatal(err) + } + } + msg := VoteMessage{Vote: NewSkipVote(5000), Rank: 1} + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if _, err := o.ObserveVote(msg); err != nil { + b.Fatal(err) + } + } +} + +// Model a full diagnostic history, with a small unresolved frontier or an +// entirely unresolved history as a worst case. These are observer-only costs: +// no transactions, BLS verification, or network traffic are included. +func BenchmarkObserverEmptyReplay(b *testing.B) { + for _, pending := range []int{0, 32, DefaultMaxTrackedCertificates} { + for _, skips := range []int{0, 4} { + b.Run(fmt.Sprintf("pending=%d/skips=%d", pending, skips), func(b *testing.B) { + o := NewObserver() + for slot := uint64(1); slot <= DefaultMaxTrackedCertificates; slot++ { + if slot <= uint64(DefaultMaxTrackedCertificates-pending) { + o.ObserveReplayBlock(ReplayBlockObservation{Block: BlockID{Slot: slot, Hash: testHash(1)}}) + } + _, err := o.ObserveCertificate(Certificate{Type: CertificateNotarize, Slot: slot, + BlockHash: testHash(1), IncludedStake: 80, TotalStake: 100}) + if err != nil { + b.Fatal(err) + } + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + base := uint64(DefaultMaxTrackedCertificates + 1 + i*(skips+1)) + for n := 0; n < skips; n++ { + o.ObserveReplayBlock(ReplayBlockObservation{Block: BlockID{Slot: base + uint64(n)}}) + } + o.ObserveReplayBlock(ReplayBlockObservation{Block: BlockID{Slot: base + uint64(skips), Hash: testHash(1)}}) + } + }) + } + } +} diff --git a/pkg/alpenglow/observer_stats_test.go b/pkg/alpenglow/observer_stats_test.go new file mode 100644 index 000000000..bf81999de --- /dev/null +++ b/pkg/alpenglow/observer_stats_test.go @@ -0,0 +1,127 @@ +package alpenglow + +import ( + "fmt" + "math/rand" + "sync" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestObserverPendingStatsMatchFullScan(t *testing.T) { + // Exercise eviction, duplicate certificates, out-of-order and hashless + // replay, and retained matches/mismatches with and without retention. + for _, retention := range []int{0, 1, 8, 64} { + t.Run(fmt.Sprint(retention), func(t *testing.T) { + o := NewObserverWithConfig(ObserverConfig{MaxTrackedVotes: 8, + MaxTrackedCertificates: retention, MaxTrackedReplayBlocks: retention}) + rng := rand.New(rand.NewSource(37)) + for i := 0; i < 1000; i++ { + slot := uint64(rng.Intn(80) + 1) + switch rng.Intn(5) { + case 0, 1: + typ := CertificateNotarize + if i%3 == 0 { + typ = CertificateFinalizeFast + } + _, err := o.ObserveCertificate(Certificate{Type: typ, Slot: slot, + BlockHash: testHash(byte(slot%3 + 1)), IncludedStake: 80, TotalStake: 100}) + require.NoError(t, err) + case 2: + block := BlockID{Slot: slot} + if i%5 != 0 { + block.Hash = testHash(byte(slot%4 + 1)) + } + o.ObserveReplayBlock(ReplayBlockObservation{Block: block}) + case 3: + _, err := o.ObserveVote(VoteMessage{Vote: NewSkipVote(slot), Rank: 1}) + require.NoError(t, err) + case 4: + o.ObserveReplayResult(ReplayResultObservation{Slot: slot}) + } + o.Snapshot() + o.mu.RLock() + cached, scanned := o.pendingStats, observerPendingStatsFullScan(o) + pending := make(map[CertificateKey]BlockID) + for key, cert := range o.certificates { + if _, checked := o.replayChecks[key]; !checked { + if block, ok := cert.Block(); ok && block.HasHash() { + pending[key] = block + } + } + } + require.Equal(t, pending, o.pendingReplayCertificates, "operation %d", i) + o.mu.RUnlock() + require.Equal(t, scanned, cached, "operation %d", i) + } + }) + } +} + +func TestObserverConcurrentSnapshotsAndReconciliation(t *testing.T) { + o := NewObserver() + var wg sync.WaitGroup + for worker := 0; worker < 4; worker++ { + wg.Add(1) + go func(worker int) { + defer wg.Done() + for slot := uint64(1); slot <= 100; slot++ { + switch worker { + case 0: + o.ObserveReplayBlock(ReplayBlockObservation{Block: BlockID{Slot: slot, Hash: testHash(1)}}) + case 1: + _, err := o.ObserveCertificate(Certificate{Type: CertificateNotarize, Slot: slot, + BlockHash: testHash(1), IncludedStake: 80, TotalStake: 100}) + if err != nil { + t.Error(err) + } + case 2: + _, err := o.ObserveVote(VoteMessage{Vote: NewSkipVote(slot), Rank: 1}) + if err != nil { + t.Error(err) + } + case 3: + o.Snapshot() + } + } + }(worker) + } + wg.Wait() + snapshot := o.Snapshot() + require.Equal(t, uint64(100), snapshot.CertificateReplayMatches) + require.Zero(t, snapshot.CertificateReplayPending) +} + +// Independent reference retains the original scan over every retained certificate. +func observerPendingStatsFullScan(o *Observer) certificateReplayPendingStats { + var stats certificateReplayPendingStats + for key, cert := range o.certificates { + if _, checked := o.replayChecks[key]; checked { + continue + } + certBlock, ok := cert.Block() + if ok && certBlock.HasHash() { + stats.count++ + if stats.oldestSlot == 0 || certBlock.Slot < stats.oldestSlot { + stats.oldestSlot = certBlock.Slot + } + if certBlock.Slot > stats.newestSlot { + stats.newestSlot = certBlock.Slot + } + if o.oldestReplayBlockSlot != 0 && certBlock.Slot < o.oldestReplayBlockSlot { + stats.preWindow++ + } + if o.oldestReplayBlockSlot != 0 && + o.latestReplayBlockSlot != 0 && + certBlock.Slot >= o.oldestReplayBlockSlot && + certBlock.Slot <= o.latestReplayBlockSlot { + stats.mature++ + if stats.matureOldestSlot == 0 || certBlock.Slot < stats.matureOldestSlot { + stats.matureOldestSlot = certBlock.Slot + } + } + } + } + return stats +} diff --git a/pkg/alpenglow/peer_sender.go b/pkg/alpenglow/peer_sender.go new file mode 100644 index 000000000..73c9176f6 --- /dev/null +++ b/pkg/alpenglow/peer_sender.go @@ -0,0 +1,181 @@ +package alpenglow + +import ( + "errors" + "sync" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/quic-go/quic-go" +) + +const ( + defaultVotorPeerSendQueue = 256 + votorSendTimeout = time.Second + votorSendWatchInterval = 100 * time.Millisecond +) + +// VotorPeerQueueStats describes local queueing, not remote delivery. Counters +// here cover the current connection; broadcaster totals survive reconnects. +type VotorPeerQueueStats struct { + Identity solana.PublicKey `json:"identity"` + Address string `json:"address"` + Queued int `json:"queued"` + SendingFor time.Duration `json:"sending_for_ns"` + LastQueueDelay time.Duration `json:"last_queue_delay_ns"` + MaxQueueDelay time.Duration `json:"max_queue_delay_ns"` + QueueDrops uint64 `json:"queue_drops"` +} + +type votorDatagram struct { + payload []byte // immutable, shared across the peer queues + queuedAt time.Time +} + +// Each authenticated connection owns one sender. No mutex is held while +// SendDatagram blocks on quic-go's bounded datagram queue. +type votorPeerSender struct { + b *VotorBroadcaster + peer VotorPeer + conn *quic.Conn + queue chan votorDatagram + done chan struct{} + + mu sync.Mutex + closed bool + sendingSince time.Time + lastQueueDelay time.Duration + maxQueueDelay time.Duration + queueDrops uint64 +} + +func (s *votorPeerSender) enqueue(job votorDatagram) { + s.mu.Lock() + defer s.mu.Unlock() + if s.closed || s.conn.Context().Err() != nil { + s.b.sendsSkipped.Add(1) + return + } + select { + case s.queue <- job: + default: + // A full queue rejects only this peer's copy. As before, fanout is + // best effort; the global Enqueue error contract is unchanged. + s.queueDrops++ + s.b.peerQueueDrops.Add(1) + s.b.dropped.Add(1) + } +} + +func (s *votorPeerSender) run() { + defer s.b.wg.Done() + defer close(s.done) + // Cover both remote-close select and close detected after dequeuing a job. + // The queue is bounded/deduplicated; shutdown, departure and a healthy + // replacement connection suppress obsolete reconnect requests. + defer s.b.queueConnect(s.peer.Identity) + defer func() { + s.mu.Lock() + s.closed = true + // Old-connection work is not replayed onto a new address/connection. + // Account for every queued copy discarded on failure or shutdown. + for { + select { + case <-s.queue: + s.b.peerQueueDiscarded.Add(1) + default: + s.mu.Unlock() + return + } + } + }() + for { + select { + case <-s.b.done: + return + case <-s.conn.Context().Done(): + return + case job := <-s.queue: + s.mu.Lock() + if s.closed || s.conn.Context().Err() != nil || s.b.closed.Load() { + s.mu.Unlock() + s.b.peerQueueDiscarded.Add(1) + return + } + now := time.Now() + s.sendingSince = now + s.lastQueueDelay = now.Sub(job.queuedAt) + s.maxQueueDelay = max(s.maxQueueDelay, s.lastQueueDelay) + for old := s.b.peerQueueMaxDelay.Load(); int64(s.lastQueueDelay) > old; old = s.b.peerQueueMaxDelay.Load() { + if s.b.peerQueueMaxDelay.CompareAndSwap(old, int64(s.lastQueueDelay)) { + break + } + } + if s.lastQueueDelay >= votorSendTimeout { + // Do not feed an already-stale backlog into a briefly writable + // QUIC queue between watchdog ticks. Retire this connection. + s.closed = true + s.mu.Unlock() + s.b.peerQueueDiscarded.Add(1) + s.timeout() + return + } + s.mu.Unlock() + err := s.conn.SendDatagram(job.payload) + s.mu.Lock() + s.sendingSince = time.Time{} + s.mu.Unlock() + if err != nil { + s.b.recordSendError(s.peer, err) + var tooLarge *quic.DatagramTooLargeError + if errors.As(err, &tooLarge) { + continue + } + s.b.dropConnection(s.peer.Identity, s.conn) + s.b.queueConnect(s.peer.Identity) + return + } + s.b.sends.Add(1) + } + } +} + +func (s *votorPeerSender) stats() VotorPeerQueueStats { + s.mu.Lock() + defer s.mu.Unlock() + var sendingFor time.Duration + if !s.sendingSince.IsZero() { + sendingFor = time.Since(s.sendingSince) + } + return VotorPeerQueueStats{ + Identity: s.peer.Identity, Address: s.peer.Addr.String(), + Queued: len(s.queue), SendingFor: sendingFor, + LastQueueDelay: s.lastQueueDelay, MaxQueueDelay: s.maxQueueDelay, QueueDrops: s.queueDrops, + } +} + +func (b *VotorBroadcaster) expirePeerSends(now time.Time) { + senders, _ := b.connectedSenders() + for _, s := range senders { + s.mu.Lock() + expired := !s.closed && !s.sendingSince.IsZero() && s.lastQueueDelay+now.Sub(s.sendingSince) >= votorSendTimeout + if expired { + // Serialize with send completion so a late watchdog cannot close a + // later, unrelated send after the blocked operation has finished. + s.closed = true + } + s.mu.Unlock() + if expired { + s.timeout() + } + } +} + +// Caller must first claim the timeout by setting closed under s.mu. This makes +// the dequeue check and watchdog mutually exclusive and counts one timeout. +func (s *votorPeerSender) timeout() { + s.b.peerSendTimeouts.Add(1) + // Closing wakes SendDatagram without leaking a timeout goroutine. + s.b.dropConnection(s.peer.Identity, s.conn) + s.b.queueConnect(s.peer.Identity) +} diff --git a/pkg/alpenglow/peer_sender_recovery_test.go b/pkg/alpenglow/peer_sender_recovery_test.go new file mode 100644 index 000000000..3f88446ea --- /dev/null +++ b/pkg/alpenglow/peer_sender_recovery_test.go @@ -0,0 +1,136 @@ +package alpenglow + +import ( + "context" + "crypto/ed25519" + "crypto/tls" + "net" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Real authenticated connection with no reconciliation loop or connection +// workers. Any reconnect job must come from the sender, never the periodic tick. +func passiveVotorSender(t *testing.T) (*VotorBroadcaster, *votorPeerSender, *Receiver) { + t.Helper() + serverIdentity := ed25519.NewKeyFromSeed(bytesOf(181, ed25519.SeedSize)) + r, err := NewReceiver(ReceiverConfig{ + BindAddr: "127.0.0.1:0", Identity: serverIdentity, LogInterval: -1, + AdmitPeer: func(solana.PublicKey) bool { return true }, + }, NewObserver()) + require.NoError(t, err) + runVotorReceiver(t, r) + cert, err := newVotorQUICCertificate(ed25519.NewKeyFromSeed(bytesOf(182, ed25519.SeedSize))) + require.NoError(t, err) + ctx, cancel := context.WithCancel(context.Background()) + peer := VotorPeer{Identity: testVotorPubkey(serverIdentity), Addr: r.Addr().(*net.UDPAddr)} + b := &VotorBroadcaster{ + ctx: ctx, cancel: cancel, done: make(chan struct{}), + tlsConfig: &tls.Config{Certificates: []tls.Certificate{cert}, NextProtos: []string{VotorQUICALPN}, MinVersion: tls.VersionTLS13, InsecureSkipVerify: true}, + quicConfig: newVotorQUICConfig(), + jobs: make(chan votorPeerJob, 2), desired: map[solana.PublicKey]VotorPeer{peer.Identity: peer}, + conns: make(map[solana.PublicKey]votorConnection), dialing: make(map[solana.PublicKey]*votorDial), connectQueued: make(map[solana.PublicKey]struct{}), + } + t.Cleanup(func() { require.NoError(t, b.Close()) }) + _, err = b.connection(peer) + require.NoError(t, err) + b.connMu.Lock() + sender := b.conns[peer.Identity].sender + b.connMu.Unlock() + return b, sender, r +} + +func TestVotorPeerReconnectOnRemoteCloseWithoutReconcile(t *testing.T) { + for _, duringDequeue := range []bool{false, true} { + name := "idle" + if duringDequeue { + name = "dequeue" + } + t.Run(name, func(t *testing.T) { + b, s, receiver := passiveVotorSender(t) + if duringDequeue { + func() { + s.mu.Lock() + defer s.mu.Unlock() + s.queue <- votorDatagram{payload: []byte{1}, queuedAt: time.Now()} + // Force the job branch to win select, then close remotely + // while the sender is waiting to check the connection state. + require.Eventually(t, func() bool { return len(s.queue) == 0 }, time.Second, time.Millisecond) + require.NoError(t, receiver.Close()) + select { + case <-s.conn.Context().Done(): + case <-time.After(time.Second): + t.Fatal("remote close not observed") + } + }() + } else { + require.NoError(t, receiver.Close()) + } + select { + case <-s.done: + case <-time.After(time.Second): + t.Fatal("sender did not exit") + } + select { + case job := <-b.jobs: + require.Equal(t, s.peer.Identity, job.peer.Identity) + default: + t.Fatal("sender exited without requesting reconnect") + } + require.Empty(t, b.jobs, "only one reconnect request per peer") + }) + } +} + +func TestVotorPeerDeadlineIncludesQueueAge(t *testing.T) { + for _, tc := range []struct { + name string + queued, sending time.Duration + active, expired bool + }{ + {"progress_does_not_reset_age", 950 * time.Millisecond, 100 * time.Millisecond, true, true}, + {"below_deadline", 800 * time.Millisecond, 100 * time.Millisecond, true, false}, + {"blocked_call", 0, time.Second, true, true}, + {"completed_send", 2 * time.Second, 0, false, false}, + } { + t.Run(tc.name, func(t *testing.T) { + b, s, _ := passiveVotorSender(t) + now := time.Now() + s.mu.Lock() + if tc.active { + s.sendingSince = now.Add(-tc.sending) + } + s.lastQueueDelay = tc.queued + s.mu.Unlock() + // Inject a deterministic clock/state boundary. There is no actual + // datagram in progress and no timer goroutine in this fixture. + b.expirePeerSends(now) + require.Equal(t, tc.expired, s.conn.Context().Err() != nil) + if tc.expired { + require.EqualValues(t, 1, b.peerSendTimeouts.Load()) + b.expirePeerSends(now.Add(time.Second)) + require.EqualValues(t, 1, b.peerSendTimeouts.Load()) + } else { + require.Zero(t, b.peerSendTimeouts.Load()) + } + }) + } +} + +func TestVotorPeerRejectsAgedQueueBeforeQUICEnqueue(t *testing.T) { + b, s, _ := passiveVotorSender(t) + s.enqueue(votorDatagram{payload: []byte{1}, queuedAt: time.Now().Add(-2 * time.Second)}) + select { + case <-s.done: + case <-time.After(time.Second): + t.Fatal("aged queue did not retire its connection") + } + require.Error(t, s.conn.Context().Err()) + require.Zero(t, b.sends.Load(), "stale backlog must not enter the QUIC queue") + require.EqualValues(t, 1, b.peerSendTimeouts.Load()) + require.EqualValues(t, 1, b.peerQueueDiscarded.Load()) + require.Len(t, b.jobs, 1) +} diff --git a/pkg/alpenglow/peer_sender_test.go b/pkg/alpenglow/peer_sender_test.go new file mode 100644 index 000000000..7595643fe --- /dev/null +++ b/pkg/alpenglow/peer_sender_test.go @@ -0,0 +1,249 @@ +package alpenglow + +import ( + "crypto/ed25519" + "net" + "runtime" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Real QUIC traffic: one path is blackholed after authentication while the +// other stays healthy. This reproduces the original shared-worker failure. +func TestVotorBroadcasterIsolatesBlockedPeer(t *testing.T) { + for _, action := range []string{"reconnect", "close", "depart", "move"} { + t.Run(action, func(t *testing.T) { testVotorBlockedPeer(t, action) }) + } +} + +func testVotorBlockedPeer(t *testing.T, action string) { + const markerSlot = 999999 + marker := make(chan time.Time, 1) + badMarker := make(chan time.Time, 1) + makeReceiver := func(seed byte, record bool) (*Receiver, solana.PublicKey) { + identity := ed25519.NewKeyFromSeed(bytesOf(seed, ed25519.SeedSize)) + r, err := NewReceiver(ReceiverConfig{ + BindAddr: "127.0.0.1:0", Identity: identity, LogInterval: -1, + MaxDatagramsPerSecond: 100000, + AdmitPeer: func(solana.PublicKey) bool { return true }, + AdmitMessage: func(_ solana.PublicKey, m Message) (Message, bool) { + if m.Slot() == markerSlot { + target := badMarker + if record { + target = marker + } + select { + case target <- time.Now(): + default: + } + } + return Message{}, false + }, + }, NewObserver()) + require.NoError(t, err) + runVotorReceiver(t, r) + return r, testVotorPubkey(identity) + } + badReceiver, badID := makeReceiver(191, false) + goodReceiver, goodID := makeReceiver(192, true) + badAddr := badReceiver.Addr().(*net.UDPAddr) + proxy, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + var blackhole atomic.Bool + proxyDone := make(chan struct{}) + go func() { + defer close(proxyDone) + buf := make([]byte, 65536) + var client *net.UDPAddr + for { + n, from, err := proxy.ReadFromUDP(buf) + if err != nil { + return + } + if blackhole.Load() { + continue + } + if from.String() == badAddr.String() { + if client != nil { + _, _ = proxy.WriteToUDP(buf[:n], client) + } + } else { + client = from + _, _ = proxy.WriteToUDP(buf[:n], badAddr) + } + } + }() + t.Cleanup(func() { _ = proxy.Close(); <-proxyDone }) + badPeer := VotorPeer{Identity: badID, Addr: proxy.LocalAddr().(*net.UDPAddr)} + goodPeer := VotorPeer{Identity: goodID, Addr: goodReceiver.Addr().(*net.UDPAddr)} + peers := newMutableVotorPeers([]VotorPeer{badPeer, goodPeer}) + b, err := NewVotorBroadcaster(VotorBroadcasterConfig{ + Identity: ed25519.NewKeyFromSeed(bytesOf(193, ed25519.SeedSize)), + Peers: peers.Snapshot, + Workers: defaultVotorConnectWorkers, + }) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, b.Close()) }) + require.Eventually(t, func() bool { return b.Stats().Connections == 2 }, 3*time.Second, 5*time.Millisecond) + badConn, ok := b.establishedConnection(badPeer) + require.True(t, ok) + // Establish an actual healthy delivery before introducing the fault. + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x41), 3))) + select { + case <-marker: + case <-time.After(time.Second): + t.Fatal("healthy baseline failed") + } + select { + case <-badMarker: + case <-time.After(time.Second): + t.Fatal("proxied baseline failed") + } + blackholedAt := time.Now() + blackhole.Store(true) + for slot := uint64(1); slot <= 256; slot++ { + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(slot), testSignatureSeq(0x41), 3))) + } + blockedWorkers := func() int { + buf := make([]byte, 2<<20) + stack := string(buf[:runtime.Stack(buf, true)]) + count := 0 + for _, goroutine := range strings.Split(stack, "\n\n") { + if strings.Contains(goroutine, "(*datagramQueue).Add") && strings.Contains(goroutine, "(*votorPeerSender).run") { + count++ + } + } + return count + } + require.Eventually(t, func() bool { return blockedWorkers() == 1 }, 3*time.Second, 5*time.Millisecond) + before := b.Stats() + require.Zero(t, before.MessagesDropped) + goodConn, ok := b.establishedConnection(goodPeer) + require.True(t, ok) + b.connMu.Lock() + badSender := b.conns[badID].sender + b.connMu.Unlock() + start := time.Now() + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x42), 3))) + select { + case received := <-marker: + t.Logf("Healthy marker received in %s while other peer's SendDatagram is blocked", received.Sub(start)) + case <-badConn.Context().Done(): + t.Fatal("healthy marker did not arrive before stalled peer was released") + case <-time.After(3 * time.Second): + t.Fatal("stalled peer delayed healthy delivery") + } + require.NoError(t, badConn.Context().Err(), "marker must arrive before watchdog releases stalled peer") + if action != "reconnect" { + switch action { + case "close": + closed := make(chan struct{}) + go func() { _ = b.Close(); close(closed) }() + select { + case <-closed: + case <-time.After(3 * time.Second): + t.Fatal("Close waited for stalled SendDatagram") + } + case "depart": + peers.Set([]VotorPeer{goodPeer}) + b.reconcilePeers() + case "move": + replacement, _ := makeReceiver(191, false) + badPeer.Addr = replacement.Addr().(*net.UDPAddr) + peers.Set([]VotorPeer{badPeer, goodPeer}) + b.reconcilePeers() + } + select { + case <-badSender.done: + case <-time.After(3 * time.Second): + t.Fatal("old sender did not stop") + } + require.Error(t, badConn.Context().Err()) + require.Empty(t, badSender.queue) + require.Positive(t, b.Stats().PeerQueueDiscarded) + if action == "close" { + return + } + if action == "move" { + require.Eventually(t, func() bool { + conn, ok := b.establishedConnection(badPeer) + return ok && conn != badConn + }, 3*time.Second, 5*time.Millisecond) + } + currentGood, ok := b.establishedConnection(goodPeer) + require.True(t, ok) + require.Same(t, goodConn, currentGood) + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x44), 3))) + select { + case <-marker: + case <-time.After(time.Second): + t.Fatal("healthy delivery stopped after peer change") + } + if action == "move" { + select { + case <-badMarker: + case <-time.After(time.Second): + t.Fatal("replacement address did not receive new vote") + } + } + return + } + // Deliberately fill just the stalled peer's bounded queue. Healthy peers + // remain independent even when the failed peer's copies are rejected. + payload, err := EncodeMessage(NewVoteMessage(NewSkipVote(123), testSignatureSeq(0x42), 3)) + require.NoError(t, err) + for range 2 * defaultVotorPeerSendQueue { + badSender.enqueue(votorDatagram{payload: payload, queuedAt: time.Now()}) + } + require.Equal(t, defaultVotorPeerSendQueue, len(badSender.queue)) + require.Positive(t, b.Stats().PeerQueueDrops) + require.Equal(t, b.Stats().PeerQueueDrops, b.Stats().MessagesDropped) + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x43), 3))) + select { + case <-marker: + case <-badConn.Context().Done(): + t.Fatal("healthy marker did not arrive before stalled peer was released") + case <-time.After(3 * time.Second): + t.Fatal("full peer queue delayed healthy delivery") + } + // Queue age is measured from fanout, so PTO progress no longer restarts + // the one-second budget. Allow two seconds of test scheduling slack beyond + // one watchdog tick, measured from the fault rather than this assertion. + remaining := time.Until(blackholedAt.Add(votorSendTimeout + votorSendWatchInterval + 2*time.Second)) + require.Positive(t, remaining) + require.Eventually(t, func() bool { return badConn.Context().Err() != nil }, remaining, 5*time.Millisecond) + t.Logf("Blackholed peer retired after %s", time.Since(blackholedAt)) + select { + case <-badSender.done: + case <-time.After(time.Second): + t.Fatal("stalled sender leaked after watchdog closed its connection") + } + require.EqualValues(t, 1, b.Stats().PeerSendTimeouts) + require.Positive(t, b.Stats().PeerQueueDiscarded) + require.Empty(t, badSender.queue) + // The old queue must never be resurrected on a replacement connection. + blackhole.Store(false) + require.Eventually(t, func() bool { + conn, ok := b.establishedConnection(badPeer) + return ok && conn != badConn + }, 5*time.Second, 10*time.Millisecond) + currentGood, ok := b.establishedConnection(goodPeer) + require.True(t, ok) + require.Same(t, goodConn, currentGood) + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x44), 3))) + select { + case <-marker: + case <-time.After(time.Second): + t.Fatal("healthy peer did not continue after reconnect") + } + select { + case <-badMarker: + case <-time.After(time.Second): + t.Fatal("reconnected peer did not receive new vote") + } +} diff --git a/pkg/alpenglow/testdata/README.md b/pkg/alpenglow/testdata/README.md index 92b75d4eb..10997bf71 100644 --- a/pkg/alpenglow/testdata/README.md +++ b/pkg/alpenglow/testdata/README.md @@ -12,3 +12,12 @@ The v4.3 vote wire message intentionally excludes validator rank and stake; the authenticated Votor transport identity supplies those values after decode. The shred version is the final little-endian `u16`. Certificate bitmap vectors use wincode's default bincode-compatible little-endian `u64` length. + +`agave_votor_certificate.der` is the 249-byte certificate constructed by +`anza-xyz/agave` commit `8fe3f1201abc5b0244540aed0c7bf8c6bcafb3f5`, +`tls-utils/src/tls_certificates.rs::new_dummy_x509_certificate`, with public +key bytes `00..1f` at offsets 100–131. It also matches Firedancer commit +`de039cd7fc9f4714782ec7e3d47db3728903abc6`, +`src/ballet/x509/fd_x509_mock.c::fd_x509_mock_pubkey_v2`. Its fixed dummy +X.509 signature is intentional; TLS CertificateVerify proves possession of +the identity key. This fixture contains no private key. diff --git a/pkg/alpenglow/testdata/agave_votor_certificate.der b/pkg/alpenglow/testdata/agave_votor_certificate.der new file mode 100644 index 000000000..e23bc0bfb Binary files /dev/null and b/pkg/alpenglow/testdata/agave_votor_certificate.der differ diff --git a/pkg/alpenglow/tls_identity.go b/pkg/alpenglow/tls_identity.go index 1e8b91d91..3c18e0cfd 100644 --- a/pkg/alpenglow/tls_identity.go +++ b/pkg/alpenglow/tls_identity.go @@ -5,9 +5,7 @@ import ( "crypto/rand" "crypto/tls" "crypto/x509" - "crypto/x509/pkix" "fmt" - "math/big" "time" "github.com/gagliardetto/solana-go" @@ -27,6 +25,36 @@ func newVotorQUICConfig() *quic.Config { } } +// votorCertificateTemplate is Agave's dummy X.509 certificate from +// tls-utils/src/tls_certificates.rs (8fe3f1201abc5b0244540aed0c7bf8c6bcafb3f5). +// Firedancer's fd_x509_mock_pubkey_v2 requires this exact encoding except for +// the 32-byte SubjectPublicKeyInfo at offset 100. The X.509 signature is +// deliberately invalid: TLS 1.3 CertificateVerify authenticates the identity. +// Keep the template unchanged and copy it before inserting a public key. +var votorCertificateTemplate = [...]byte{ + 0x30, 0x81, 0xf6, 0x30, 0x81, 0xa9, 0xa0, 0x03, 0x02, 0x01, 0x02, 0x02, + 0x08, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x30, 0x05, 0x06, + 0x03, 0x2b, 0x65, 0x70, 0x30, 0x16, 0x31, 0x14, 0x30, 0x12, 0x06, 0x03, + 0x55, 0x04, 0x03, 0x0c, 0x0b, 0x53, 0x6f, 0x6c, 0x61, 0x6e, 0x61, 0x20, + 0x6e, 0x6f, 0x64, 0x65, 0x30, 0x20, 0x17, 0x0d, 0x37, 0x30, 0x30, 0x31, + 0x30, 0x31, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x5a, 0x18, 0x0f, 0x34, + 0x30, 0x39, 0x36, 0x30, 0x31, 0x30, 0x31, 0x30, 0x30, 0x30, 0x30, 0x30, + 0x30, 0x5a, 0x30, 0x00, 0x30, 0x2a, 0x30, 0x05, 0x06, 0x03, 0x2b, 0x65, + 0x70, 0x03, 0x21, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0xa3, 0x29, 0x30, 0x27, 0x30, 0x17, 0x06, 0x03, 0x55, 0x1d, 0x11, 0x01, + 0x01, 0xff, 0x04, 0x0d, 0x30, 0x0b, 0x82, 0x09, 0x6c, 0x6f, 0x63, 0x61, + 0x6c, 0x68, 0x6f, 0x73, 0x74, 0x30, 0x0c, 0x06, 0x03, 0x55, 0x1d, 0x13, + 0x01, 0x01, 0xff, 0x04, 0x02, 0x30, 0x00, 0x30, 0x05, 0x06, 0x03, 0x2b, + 0x65, 0x70, 0x03, 0x41, 0x00, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, +} + // newVotorQUICCertificate creates the single-certificate Ed25519 identity // chain used by the Agave Votor transport. Peers recover the validator identity // directly from the leaf certificate's SubjectPublicKeyInfo; TLS 1.3's @@ -47,23 +75,8 @@ func newVotorQUICCertificate(identity ed25519.PrivateKey) (tls.Certificate, erro pub = priv.Public().(ed25519.PublicKey) } - template := &x509.Certificate{ - SerialNumber: big.NewInt(1), - Subject: pkix.Name{ - CommonName: "Mithril Alpenglow observer", - }, - NotBefore: time.Unix(0, 0), - NotAfter: time.Date(4096, 1, 1, 0, 0, 0, 0, time.UTC), - KeyUsage: x509.KeyUsageDigitalSignature, - ExtKeyUsage: []x509.ExtKeyUsage{x509.ExtKeyUsageServerAuth, x509.ExtKeyUsageClientAuth}, - DNSNames: []string{"localhost"}, - BasicConstraintsValid: true, - } - - certDER, err := x509.CreateCertificate(rand.Reader, template, template, pub, priv) - if err != nil { - return tls.Certificate{}, err - } + certDER := append([]byte(nil), votorCertificateTemplate[:]...) + copy(certDER[100:100+ed25519.PublicKeySize], pub) cert, err := x509.ParseCertificate(certDER) if err != nil { return tls.Certificate{}, err diff --git a/pkg/alpenglow/tls_identity_test.go b/pkg/alpenglow/tls_identity_test.go index 7efaa4125..cb591e0fa 100644 --- a/pkg/alpenglow/tls_identity_test.go +++ b/pkg/alpenglow/tls_identity_test.go @@ -1,15 +1,61 @@ package alpenglow import ( + "context" "crypto/ed25519" "crypto/tls" "crypto/x509" + "net" + "os" "testing" "time" "github.com/stretchr/testify/require" ) +func TestVotorCertificateMatchesAgaveAndFiredancerTemplate(t *testing.T) { + // Independent fixture from Agave's constructor, also accepted by + // Firedancer's fd_x509_mock_pubkey_v2 byte-pattern parser. + fixture, err := os.ReadFile("testdata/agave_votor_certificate.der") + require.NoError(t, err) + require.Len(t, fixture, 249) + for _, seed := range []byte{61, 62} { + identity := ed25519.NewKeyFromSeed(bytesOf(seed, ed25519.SeedSize)) + certificate, err := newVotorQUICCertificate(identity) + require.NoError(t, err) + expected := append([]byte(nil), fixture...) + copy(expected[100:132], identity.Public().(ed25519.PublicKey)) + require.Equal(t, expected, certificate.Certificate[0]) + require.Equal(t, identity.Public(), certificate.Leaf.PublicKey) + // Do not replace the dummy signature with a real one: Firedancer's + // parser matches it too. Authentication happens in CertificateVerify. + require.Error(t, certificate.Leaf.CheckSignature(certificate.Leaf.SignatureAlgorithm, + certificate.Leaf.RawTBSCertificate, certificate.Leaf.Signature)) + } +} + +func TestVotorCertificateStillRequiresIdentityKeyPossession(t *testing.T) { + identity := ed25519.NewKeyFromSeed(bytesOf(63, ed25519.SeedSize)) + certificate, err := newVotorQUICCertificate(identity) + require.NoError(t, err) + certificate.PrivateKey = ed25519.NewKeyFromSeed(bytesOf(64, ed25519.SeedSize)) + serverConn, clientConn := net.Pipe() + t.Cleanup(func() { _ = serverConn.Close(); _ = clientConn.Close() }) + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + server := tls.Server(serverConn, &tls.Config{ + Certificates: []tls.Certificate{certificate}, MinVersion: tls.VersionTLS13, + }) + client := tls.Client(clientConn, &tls.Config{ + InsecureSkipVerify: true, MinVersion: tls.VersionTLS13, + }) + serverDone := make(chan error, 1) + go func() { serverDone <- server.HandshakeContext(ctx) }() + // Skipping the dummy X.509 signature does not bypass CertificateVerify. + require.ErrorContains(t, client.HandshakeContext(ctx), "invalid signature") + require.Error(t, <-serverDone) +} + func TestVotorPeerIdentityRequiresOneEd25519Certificate(t *testing.T) { identity := ed25519.NewKeyFromSeed(bytesOf(61, ed25519.SeedSize)) certificate, err := newVotorQUICCertificate(identity) diff --git a/pkg/alpenglow/vote_history.go b/pkg/alpenglow/vote_history.go index d227bbe12..448977fa3 100644 --- a/pkg/alpenglow/vote_history.go +++ b/pkg/alpenglow/vote_history.go @@ -15,6 +15,7 @@ import ( ) const voteHistoryVersion = 1 +const reservedVoteHistoryVersion = 2 var ErrVoteHistoryNotFound = errors.New("alpenglow vote history not found") @@ -23,6 +24,7 @@ var ErrVoteHistoryNotFound = errors.New("alpenglow vote history not found") // Agave when resuming (notarized blocks and ParentReady edges), in addition to // the anti-equivocation vote sets. type VoteHistory struct { + ReservationRequired bool `json:"reservation_required,omitempty"` Version uint32 `json:"version"` NodePubkey solana.PublicKey `json:"node_pubkey"` Root uint64 `json:"root"` @@ -420,50 +422,111 @@ func VoteHistoryFilename(dir string, node solana.PublicKey) string { return filepath.Join(dir, fmt.Sprintf("vote_history-%s.mithril.json", node)) } -// SaveVoteHistory signs the exact serialized history with the validator -// identity and atomically replaces the previous file before a vote can be -// admitted to consensus or sent to the network. +// SaveVoteHistory authenticates and durably replaces the exact history: write, +// file sync, rename, then directory sync. Synchronous-mode callers require +// success before pool admission (which can publish certificates) or network +// enqueue. The BLS signature may already have been computed privately in RAM; +// this is persist-before-publication, not persist-before-BLS-computation. func SaveVoteHistory(dir string, h *VoteHistory, identity ed25519.PrivateKey) error { + return saveVoteHistory(dir, h, identity, true) +} + +// SaveReservedVoteHistory writes and renames without per-vote sync. Success +// does not prove that this history survived a host/power failure. The caller +// must enforce an independently durable signing reservation and its restart +// quarantine; a valid-looking older history is not evidence of completeness. +func SaveReservedVoteHistory(dir string, h *VoteHistory, identity ed25519.PrivateKey) error { + if h == nil || !h.ReservationRequired { + return fmt.Errorf("reserved history requires durable reservation enrollment") + } + return saveVoteHistory(dir, h, identity, false) +} + +// VoteHistorySnapshot owns signed, immutable bytes. It retains no reference to +// the voter's mutable maps or signing key and can be saved by a worker. +type VoteHistorySnapshot struct { + node solana.PublicKey + encoded []byte +} + +// PrepareReservedVoteHistory validates and signs the complete current history +// without filesystem access. Only the owner of h may call this while mutating h. +func PrepareReservedVoteHistory(h *VoteHistory, identity ed25519.PrivateKey) (*VoteHistorySnapshot, error) { + if h == nil || !h.ReservationRequired { + return nil, fmt.Errorf("reserved history requires durable reservation enrollment") + } + encoded, err := encodeVoteHistory(h, identity) + if err != nil { + return nil, err + } + return &VoteHistorySnapshot{node: h.NodePubkey, encoded: encoded}, nil +} + +// SaveReservedVoteHistorySnapshot replaces history without an explicit sync. +// Success means replacement completed, not durable vote acknowledgement. Callers +// must serialize writes and enforce the independent durable reservation. On an +// unclean restart even an intact snapshot cannot bypass the startup bound. +func SaveReservedVoteHistorySnapshot(dir string, snapshot *VoteHistorySnapshot) error { + if snapshot == nil || len(snapshot.encoded) == 0 { + return fmt.Errorf("save reserved history: empty snapshot") + } + if err := ensureDurableVoteHistoryDirectory(dir); err != nil { + return err + } + return replaceVoteHistoryFile(dir, VoteHistoryFilename(dir, snapshot.node), snapshot.encoded, false) +} + +func saveVoteHistory(dir string, h *VoteHistory, identity ed25519.PrivateKey, durable bool) error { + encoded, err := encodeVoteHistory(h, identity) + if err != nil { + return err + } + if err := ensureDurableVoteHistoryDirectory(dir); err != nil { + return err + } + return replaceVoteHistoryFile(dir, VoteHistoryFilename(dir, h.NodePubkey), encoded, durable) +} + +func encodeVoteHistory(h *VoteHistory, identity ed25519.PrivateKey) ([]byte, error) { if h == nil { - return fmt.Errorf("save vote history: nil history") + return nil, fmt.Errorf("save vote history: nil history") } if len(identity) != ed25519.PrivateKeySize { - return fmt.Errorf("save vote history: invalid identity key size %d", len(identity)) + return nil, fmt.Errorf("save vote history: invalid identity key size %d", len(identity)) } node := solana.PublicKey(identity.Public().(ed25519.PublicKey)) if node != h.NodePubkey { - return fmt.Errorf("save vote history: identity %s does not match history %s", node, h.NodePubkey) + return nil, fmt.Errorf("save vote history: identity %s does not match history %s", node, h.NodePubkey) } h.Version = voteHistoryVersion + if h.ReservationRequired { + h.Version = reservedVoteHistoryVersion + } if err := h.preparePersistedViews(); err != nil { - return fmt.Errorf("save vote history: %w", err) + return nil, fmt.Errorf("save vote history: %w", err) } defer func() { h.PersistedNotarized = nil h.PersistedParentReady = nil }() if err := h.validatePersistedState(); err != nil { - return fmt.Errorf("save vote history: %w", err) + return nil, fmt.Errorf("save vote history: %w", err) } data, err := json.Marshal(h) if err != nil { - return fmt.Errorf("serialize vote history: %w", err) + return nil, fmt.Errorf("serialize vote history: %w", err) } envelope := savedVoteHistory{ - Version: voteHistoryVersion, + Version: h.Version, Node: node, Data: data, Signature: ed25519.Sign(identity, data), } encoded, err := json.Marshal(envelope) if err != nil { - return fmt.Errorf("serialize saved vote history: %w", err) + return nil, fmt.Errorf("serialize saved vote history: %w", err) } - if err := ensureDurableVoteHistoryDirectory(dir); err != nil { - return err - } - filename := VoteHistoryFilename(dir, node) - return persistVoteHistoryFile(dir, filename, encoded) + return encoded, nil } // ensureDurableVoteHistoryDirectory creates each missing path component and @@ -515,6 +578,10 @@ func ensureDurableVoteHistoryDirectory(dir string) error { // alone does not guarantee that either the bytes or the new directory entry // survives a crash. func persistVoteHistoryFile(dir, filename string, encoded []byte) error { + return replaceVoteHistoryFile(dir, filename, encoded, true) +} + +func replaceVoteHistoryFile(dir, filename string, encoded []byte, durable bool) error { temporary, err := os.CreateTemp(dir, "."+filepath.Base(filename)+".tmp-") if err != nil { return fmt.Errorf("create temporary vote history: %w", err) @@ -538,8 +605,10 @@ func persistVoteHistoryFile(dir, filename string, encoded []byte) error { if n != len(encoded) { return fmt.Errorf("write temporary vote history: %w", io.ErrShortWrite) } - if err := temporary.Sync(); err != nil { - return fmt.Errorf("sync temporary vote history: %w", err) + if durable { + if err := temporary.Sync(); err != nil { + return fmt.Errorf("sync temporary vote history: %w", err) + } } closeErr := temporary.Close() closed = true @@ -551,8 +620,8 @@ func persistVoteHistoryFile(dir, filename string, encoded []byte) error { } renamed = true - if err := syncVoteHistoryDirectory(dir); err != nil { - return err + if durable { + return syncVoteHistoryDirectory(dir) } return nil } @@ -586,7 +655,7 @@ func LoadVoteHistory(dir string, node solana.PublicKey) (*VoteHistory, error) { if err := json.Unmarshal(encoded, &envelope); err != nil { return nil, fmt.Errorf("decode saved vote history: %w", err) } - if envelope.Version != voteHistoryVersion || envelope.Node != node { + if (envelope.Version != voteHistoryVersion && envelope.Version != reservedVoteHistoryVersion) || envelope.Node != node { return nil, fmt.Errorf("saved vote history identity/version mismatch") } if !ed25519.Verify(ed25519.PublicKey(node[:]), envelope.Data, envelope.Signature) { @@ -596,7 +665,7 @@ func LoadVoteHistory(dir string, node solana.PublicKey) (*VoteHistory, error) { if err := json.Unmarshal(envelope.Data, &h); err != nil { return nil, fmt.Errorf("decode vote history: %w", err) } - if h.Version != voteHistoryVersion || h.NodePubkey != node { + if h.Version != envelope.Version || h.NodePubkey != node || h.ReservationRequired != (h.Version == reservedVoteHistoryVersion) { return nil, fmt.Errorf("vote history identity/version mismatch") } if err := h.validatePersistedState(); err != nil { diff --git a/pkg/alpenglow/vote_history_snapshot_test.go b/pkg/alpenglow/vote_history_snapshot_test.go new file mode 100644 index 000000000..540e6fa14 --- /dev/null +++ b/pkg/alpenglow/vote_history_snapshot_test.go @@ -0,0 +1,49 @@ +package alpenglow + +import ( + "crypto/ed25519" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestReservedHistorySnapshotIsImmutable(t *testing.T) { + identity := ed25519.NewKeyFromSeed(make([]byte, ed25519.SeedSize)) + node := solana.PublicKey(identity.Public().(ed25519.PublicKey)) + h := NewVoteHistory(node, 10) + h.ReservationRequired = true + block := BlockID{Slot: 11, Hash: solana.Hash{11}} + require.NoError(t, h.AddVote(NewNotarizationVote(11, block.Hash))) + h.NotarizedBlocks[block] = true + h.AddParentReady(12, block) + snapshot, err := PrepareReservedVoteHistory(h, identity) + require.NoError(t, err) + // Mutate/prune every transport-backed collection after taking the snapshot. + h.SetRoot(20) + require.NoError(t, h.AddVote(NewSkipVote(21))) + for i := range identity { + identity[i] = 0 + } + dir := t.TempDir() + require.NoError(t, SaveReservedVoteHistorySnapshot(dir, snapshot)) + loaded, err := LoadVoteHistory(dir, node) + require.NoError(t, err) + require.Equal(t, uint64(10), loaded.Root) + require.True(t, loaded.VotedAt(11)) + require.True(t, loaded.IsBlockNotarized(block)) + require.True(t, loaded.IsParentReady(12, block)) + require.False(t, loaded.HasSkipped(21)) +} + +func TestReservedHistorySnapshotRequiresEnrollmentAndValidHistory(t *testing.T) { + identity := ed25519.NewKeyFromSeed(make([]byte, ed25519.SeedSize)) + h := NewVoteHistory(solana.PublicKey(identity.Public().(ed25519.PublicKey)), 10) + _, err := PrepareReservedVoteHistory(h, identity) + require.Error(t, err) + h.ReservationRequired = true + h.Voted[11] = true // Inconsistent with canonical VotesCast. + _, err = PrepareReservedVoteHistory(h, identity) + require.Error(t, err) + require.Error(t, SaveReservedVoteHistorySnapshot(t.TempDir(), &VoteHistorySnapshot{})) +} diff --git a/pkg/alpenglow/vote_reservation.go b/pkg/alpenglow/vote_reservation.go new file mode 100644 index 000000000..b577da189 --- /dev/null +++ b/pkg/alpenglow/vote_reservation.go @@ -0,0 +1,104 @@ +package alpenglow + +import ( + "crypto/ed25519" + "crypto/sha256" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + + "github.com/gagliardetto/solana-go" + "golang.org/x/sys/unix" +) + +// VoteReservation is the durable upper bound on slots this identity may sign, +// not a record of slots actually signed. It must survive independently of +// AccountsDB checkpoints and must never be restored from an older snapshot. +// A signature authenticates this file; it does not prove freshness against +// rollback, nor does Generation provide an external monotonic counter. +// CleanHistoryDigest permits exact-history vote recovery only after validation +// and durable consumption by a dirty successor before signing. It does not +// certify complete leader-block history. See docs/reserved-vote-history.md. +type VoteReservation struct { + Version uint32 `json:"version"` + Node solana.PublicKey `json:"node"` + VoteAccount solana.PublicKey `json:"vote_account"` + AuthorizedVoter solana.PublicKey `json:"authorized_voter"` + Genesis solana.Hash `json:"genesis"` + ShredVersion uint16 `json:"shred_version"` + Generation uint64 `json:"generation"` + Through uint64 `json:"through"` + CleanHistoryDigest []byte `json:"clean_history_digest,omitempty"` +} + +func VoteReservationFilename(dir string, node solana.PublicKey) string { + return filepath.Join(dir, fmt.Sprintf("vote_reservation-%s.mithril.json", node)) +} + +// LockVoteHistory excludes concurrent owners of this directory. Operators must +// still fence copies of the same identity on other hosts or in other paths. +func LockVoteHistory(dir string, node solana.PublicKey) (*os.File, error) { + if err := ensureDurableVoteHistoryDirectory(dir); err != nil { + return nil, err + } + f, err := os.OpenFile(filepath.Join(dir, ".vote_history-"+node.String()+".lock"), os.O_CREATE|os.O_RDWR, 0600) + if err != nil { + return nil, err + } + if err := unix.Flock(int(f.Fd()), unix.LOCK_EX|unix.LOCK_NB); err != nil { + f.Close() + return nil, fmt.Errorf("vote history already owned: %w", err) + } + return f, nil +} + +func LoadVoteReservation(dir string, node solana.PublicKey) (VoteReservation, error) { + var r VoteReservation + encoded, err := os.ReadFile(VoteReservationFilename(dir, node)) + if err != nil { + return r, err + } + var envelope savedVoteHistory + if err := json.Unmarshal(encoded, &envelope); err != nil { + return r, err + } + if envelope.Version != 1 || envelope.Node != node || !ed25519.Verify(ed25519.PublicKey(node[:]), envelope.Data, envelope.Signature) { + return r, errors.New("invalid vote reservation signature/version/identity") + } + if err := json.Unmarshal(envelope.Data, &r); err != nil { + return r, err + } + if r.Version != 1 || r.Node != node || r.Generation == 0 || (len(r.CleanHistoryDigest) != 0 && len(r.CleanHistoryDigest) != sha256.Size) { + return r, errors.New("invalid vote reservation record") + } + return r, nil +} + +func SaveVoteReservation(dir string, r VoteReservation, identity ed25519.PrivateKey) error { + if len(identity) != ed25519.PrivateKeySize || solana.PublicKey(identity.Public().(ed25519.PublicKey)) != r.Node || r.Version != 1 || r.Generation == 0 { + return errors.New("invalid vote reservation signer/record") + } + data, err := json.Marshal(r) + if err != nil { + return err + } + encoded, err := json.Marshal(savedVoteHistory{Version: 1, Node: r.Node, Data: data, Signature: ed25519.Sign(identity, data)}) + if err != nil { + return err + } + if err := ensureDurableVoteHistoryDirectory(dir); err != nil { + return err + } + return persistVoteHistoryFile(dir, VoteReservationFilename(dir, r.Node), encoded) +} + +func VoteHistoryDigest(dir string, node solana.PublicKey) ([]byte, error) { + data, err := os.ReadFile(VoteHistoryFilename(dir, node)) + if err != nil { + return nil, err + } + digest := sha256.Sum256(data) + return digest[:], nil +} diff --git a/pkg/alpenglow/vote_reservation_test.go b/pkg/alpenglow/vote_reservation_test.go new file mode 100644 index 000000000..9e1840ade --- /dev/null +++ b/pkg/alpenglow/vote_reservation_test.go @@ -0,0 +1,71 @@ +package alpenglow + +import ( + "crypto/ed25519" + "encoding/json" + "os" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestReservedHistoryFormatAndIntegrity(t *testing.T) { + key := ed25519.NewKeyFromSeed(make([]byte, 32)) + node := solana.PublicKey(key.Public().(ed25519.PublicKey)) + dir := t.TempDir() + h := NewVoteHistory(node, 39) + require.Error(t, SaveReservedVoteHistory(dir, h, key)) + h.ReservationRequired = true + require.NoError(t, h.AddVote(NewSkipVote(44))) + require.NoError(t, SaveReservedVoteHistory(dir, h, key)) + raw, err := os.ReadFile(VoteHistoryFilename(dir, node)) + require.NoError(t, err) + var envelope savedVoteHistory + require.NoError(t, json.Unmarshal(raw, &envelope)) + require.Equal(t, uint32(2), envelope.Version, "legacy reader must reject hybrid history") + loaded, err := LoadVoteHistory(dir, node) + require.NoError(t, err) + require.True(t, loaded.HasSkipped(44)) + require.True(t, loaded.ReservationRequired) + envelope.Data[10] ^= 1 + raw, err = json.Marshal(envelope) + require.NoError(t, err) + require.NoError(t, os.WriteFile(VoteHistoryFilename(dir, node), raw, 0600)) + _, err = LoadVoteHistory(dir, node) + require.Error(t, err) +} + +func BenchmarkVoteHistoryPersistence(b *testing.B) { + key := ed25519.NewKeyFromSeed(make([]byte, 32)) + node := solana.PublicKey(key.Public().(ed25519.PublicKey)) + for _, reserved := range []bool{false, true} { + name := "synchronous" + if reserved { + name = "reserved-write-rename" + } + b.Run(name, func(b *testing.B) { + dir := b.TempDir() + h := NewVoteHistory(node, 39) + h.ReservationRequired = reserved + for slot := uint64(40); slot < 72; slot++ { + if err := h.AddVote(NewNotarizationVote(slot, solana.Hash{byte(slot)})); err != nil { + b.Fatal(err) + } + } + save := SaveVoteHistory + if reserved { + save = SaveReservedVoteHistory + } + if err := SaveVoteHistory(dir, h, key); err != nil { + b.Fatal(err) + } + b.ResetTimer() + for i := 0; i < b.N; i++ { + if err := save(dir, h, key); err != nil { + b.Fatal(err) + } + } + }) + } +} diff --git a/pkg/block/block.go b/pkg/block/block.go index 23bfd1f03..459fd68d5 100644 --- a/pkg/block/block.go +++ b/pkg/block/block.go @@ -14,15 +14,30 @@ import ( "github.com/gagliardetto/solana-go/rpc" ) -// TurbineIngressTimings is the per-slot decomposition carried only by a +// TurbineIngressTimings carries per-slot observations only on a // trusted in-memory Turbine block. Durations never serialize with Block. type TurbineIngressTimings struct { ShredCollection time.Duration CompletionQueueDelay time.Duration BlockDecode time.Duration + // Completion-only parse and outstanding-signature join/verification time. TransactionParse time.Duration TransactionSigverify time.Duration ReplayAdmission time.Duration + // Early durations sum completed prefetched component work, including an + // optimistic prefix later discarded, and overlap reception and each other. + // EarlyTransactionSigverify includes queueing through future completion; + // neither early duration is CPU time or an additive pipeline wall stage. + EarlyTransactionParse time.Duration + EarlyTransactionSigverify time.Duration + // Completion wait for already-claimed background parsing/submission. + // Recorded separately from BlockDecode's active completion work. + EarlyPreparationWait time.Duration + // Only retained transactions whose verification finished by ShredFullNanos. + EarlyVerifiedTransactions uint64 + // FullToReady is wall time from full shred assembly to replay-ready completion. + // It contains completion queueing, decode and outstanding verification waits. + FullToReady time.Duration } var transactionDerivedStateInitMu sync.Mutex diff --git a/pkg/block/block_test.go b/pkg/block/block_test.go index 2ffcd3029..31ac89e61 100644 --- a/pkg/block/block_test.go +++ b/pkg/block/block_test.go @@ -11,7 +11,12 @@ func TestTransactionSignaturesVerifiedMarkerIsNotSerialized(t *testing.T) { original.MarkTransactionSignaturesVerified() admissionStart := time.Now() original.MarkTurbineReplayAdmissionStart(admissionStart) - ingress := TurbineIngressTimings{ShredCollection: 12 * time.Millisecond, TransactionSigverify: 34 * time.Millisecond} + ingress := TurbineIngressTimings{ + ShredCollection: 12 * time.Millisecond, TransactionSigverify: 34 * time.Millisecond, + EarlyTransactionParse: 2 * time.Millisecond, EarlyTransactionSigverify: 56 * time.Millisecond, + EarlyPreparationWait: 3 * time.Millisecond, + EarlyVerifiedTransactions: 80, FullToReady: 35 * time.Millisecond, + } original.MarkTurbineIngressTimings(ingress) if got, ok := original.TurbineIngressTimings(); !ok || got != ingress { t.Fatalf("ingress timings = %+v, %t; want %+v, true", got, ok, ingress) diff --git a/pkg/block/message_identity.go b/pkg/block/message_identity.go index 067ac4dec..4796b90f8 100644 --- a/pkg/block/message_identity.go +++ b/pkg/block/message_identity.go @@ -1,6 +1,12 @@ package block -import "github.com/Overclock-Validator/mithril/pkg/txstatus" +import ( + "errors" + "fmt" + + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" +) // transactionState initializes the nonserialized holder under a short global // lock. Message hashing itself is protected only by the per-block state lock, @@ -34,3 +40,64 @@ func (prepared *PreparedTransactionMessageIdentities) Identity(index int) txstat func (prepared *PreparedTransactionMessageIdentities) MatchesBlock(block *Block) bool { return block != nil && prepared.matches(block.Transactions) } + +// ErrTransactionAlreadyResolved reports a v0 transaction that already carries +// address-table resolution where an unresolved, wire-decoded object is +// required. +var ErrTransactionAlreadyResolved = errors.New("transaction address-table lookups are already resolved") + +// ExecutionCopies returns execution copies of the transactions this set is +// bound to, together with the same identities bound to those copies. +// +// Streaming replay executes a block's transactions before the block is +// complete. Execution resolves a v0 transaction's address-table lookups in +// place (solana-go's SetAddressTables refuses a second call and ResolveLookups +// appends the looked-up keys to AccountKeys), so a bank that later turns out +// not to be the block's — or the block's, on a different parent — must never +// have run the block's own objects. The copies are made here, from the +// originals this set was prepared for, so an identity can only ever be +// attached to a copy of the very transaction it was computed from: each copy +// shares the original's signatures and instructions and takes the message by +// value with its own account-key slice, which is all that resolution mutates. +// The identity itself (canonical message hash, recent blockhash) is therefore +// the original's by construction; nothing here re-authenticates the signed +// message, and callers must not treat the copies as independently verified. +// +// A bound transaction that already carries resolution is refused with +// ErrTransactionAlreadyResolved: its account keys were derived elsewhere, +// against a parent this bank cannot vouch for. +func (prepared *PreparedTransactionMessageIdentities) ExecutionCopies() ([]*solana.Transaction, *PreparedTransactionMessageIdentities, error) { + if prepared == nil || len(prepared.transactions) != len(prepared.identities) || len(prepared.transactions) != len(prepared.versions) { + return nil, nil, errors.New("prepared identities are not bound to their transactions") + } + copies := make([]*solana.Transaction, len(prepared.transactions)) + for index, tx := range prepared.transactions { + if tx == nil { + return nil, nil, fmt.Errorf("transaction %d is nil", index) + } + if tx.Message.GetVersion() == solana.MessageVersionV0 && tx.Message.IsResolved() { + return nil, nil, fmt.Errorf("transaction %d: %w", index, ErrTransactionAlreadyResolved) + } + message := tx.Message + message.AccountKeys = append(solana.PublicKeySlice(nil), tx.Message.AccountKeys...) + copies[index] = &solana.Transaction{Signatures: tx.Signatures, Message: message} + } + return copies, &PreparedTransactionMessageIdentities{ + transactions: copies, + versions: append([]solana.MessageVersion(nil), prepared.versions...), + identities: append([]txstatus.TransactionMessageIdentity(nil), prepared.identities...), + }, nil +} + +// Slice returns the prepared identities for transactions [from, to) as an +// independent prepared set bound to that sub-slice. +func (prepared *PreparedTransactionMessageIdentities) Slice(from, to int) *PreparedTransactionMessageIdentities { + if prepared == nil || from < 0 || to > len(prepared.identities) || from > to { + return nil + } + return &PreparedTransactionMessageIdentities{ + transactions: prepared.transactions[from:to:to], + versions: prepared.versions[from:to:to], + identities: prepared.identities[from:to:to], + } +} diff --git a/pkg/block/message_identity_execution_copies_test.go b/pkg/block/message_identity_execution_copies_test.go new file mode 100644 index 000000000..57c520c33 --- /dev/null +++ b/pkg/block/message_identity_execution_copies_test.go @@ -0,0 +1,95 @@ +package block + +import ( + "errors" + "testing" + + "github.com/gagliardetto/solana-go" +) + +func TestPreparedTransactionMessageIdentitiesExecutionCopies(t *testing.T) { + first, second := identityTestTransaction(1), identityTestTransaction(2) + originals := []*solana.Transaction{first, second} + prepared, err := (&Block{Transactions: originals}).PrepareTransactionMessageIdentities() + if err != nil { + t.Fatalf("prepare identities: %v", err) + } + + copies, rebound, err := prepared.ExecutionCopies() + if err != nil { + t.Fatalf("execution copies: %v", err) + } + if len(copies) != 2 || copies[0] == first || copies[1] == second { + t.Fatalf("copies must be distinct objects: %v", copies) + } + for i, tx := range copies { + if &tx.Message.AccountKeys[0] == &originals[i].Message.AccountKeys[0] { + t.Fatalf("copy %d shares the original's account-key storage", i) + } + if len(tx.Signatures) != 1 || tx.Signatures[0] != originals[i].Signatures[0] { + t.Fatalf("copy %d signatures differ", i) + } + if tx.Message.RecentBlockhash != originals[i].Message.RecentBlockhash || len(tx.Message.Instructions) != 1 { + t.Fatalf("copy %d message differs", i) + } + } + if rebound.Len() != 2 || rebound.Identity(0) != prepared.Identity(0) || rebound.Identity(1) != prepared.Identity(1) { + t.Fatalf("rebound identities differ: %+v vs %+v", rebound, prepared) + } + if !rebound.matches(copies) { + t.Fatal("rebound set is not bound to the copies") + } + if rebound.matches(originals) { + t.Fatal("rebound set must not claim the originals") + } + if !prepared.matches(originals) { + t.Fatal("making copies must not alter the original set") + } + + // Resolving a v0 copy leaves the original unresolved and still resolvable. + tableID := solana.PublicKey{0x70} + lookup := identityTestTransaction(3) + lookup.Message.SetAddressTableLookups([]solana.MessageAddressTableLookup{{AccountKey: tableID, WritableIndexes: []byte{0}}}) + if _, err := lookup.Message.SetVersion(solana.MessageVersionV0); err != nil { + t.Fatalf("set v0: %v", err) + } + staticKeys := len(lookup.Message.AccountKeys) + prepared, err = (&Block{Transactions: []*solana.Transaction{lookup}}).PrepareTransactionMessageIdentities() + if err != nil { + t.Fatalf("prepare v0 identity: %v", err) + } + copies, _, err = prepared.ExecutionCopies() + if err != nil { + t.Fatalf("v0 execution copy: %v", err) + } + tables := map[solana.PublicKey]solana.PublicKeySlice{tableID: {{0x71}}} + if err := copies[0].Message.SetAddressTables(tables); err != nil { + t.Fatalf("resolve copy: %v", err) + } + if err := copies[0].Message.ResolveLookups(); err != nil { + t.Fatalf("resolve copy: %v", err) + } + if !copies[0].Message.IsResolved() || len(copies[0].Message.AccountKeys) != staticKeys+1 { + t.Fatal("copy did not resolve") + } + if lookup.Message.IsResolved() || len(lookup.Message.AccountKeys) != staticKeys { + t.Fatal("resolving the copy touched the original") + } + if err := lookup.Message.SetAddressTables(tables); err != nil { + t.Fatalf("the original must still accept its own resolution: %v", err) + } + + // An already-resolved v0 transaction is refused. + prepared, err = (&Block{Transactions: []*solana.Transaction{copies[0]}}).PrepareTransactionMessageIdentities() + if err != nil { + t.Fatalf("prepare resolved identity: %v", err) + } + if _, _, err := prepared.ExecutionCopies(); !errors.Is(err, ErrTransactionAlreadyResolved) { + t.Fatalf("resolved input must be refused, got %v", err) + } + + var none *PreparedTransactionMessageIdentities + if _, _, err := none.ExecutionCopies(); err == nil { + t.Fatal("a nil prepared set must be rejected") + } +} diff --git a/pkg/block/verified_message_identity.go b/pkg/block/verified_message_identity.go new file mode 100644 index 000000000..001067513 --- /dev/null +++ b/pkg/block/verified_message_identity.go @@ -0,0 +1,55 @@ +package block + +import ( + "fmt" + + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// PrepareVerifiedTransactionMessageIdentities binds identities from joined +// signature-verification requests to the exact ordered transaction slice they +// were computed for. Every identity must be a verified result for the +// transaction at the same index; failed/partial requests cannot seed a set. +func PrepareVerifiedTransactionMessageIdentities(transactions []*solana.Transaction, identities []txverify.VerifiedMessageIdentity) (*PreparedTransactionMessageIdentities, error) { + if len(identities) != len(transactions) { + return nil, fmt.Errorf("verified message identities do not cover transactions") + } + prepared := &PreparedTransactionMessageIdentities{ + transactions: append([]*solana.Transaction(nil), transactions...), + versions: make([]solana.MessageVersion, len(identities)), + identities: make([]txstatus.TransactionMessageIdentity, len(identities)), + } + for i, tx := range transactions { + identity, ok := identities[i].ForTransaction(tx) + if !ok { + return nil, fmt.Errorf("verified message identity does not match transaction %d", i) + } + prepared.versions[i] = tx.Message.GetVersion() + prepared.identities[i] = identity + } + return prepared, nil +} + +// CacheVerifiedTransactionMessageIdentities publishes identities from joined +// signature-verification requests. Every result must cover the exact ordered +// transaction slice. Failed/partial requests and obsolete prefetch generations +// cannot seed the cache. It does not replace block-wide duplicate/status checks. +func (b *Block) CacheVerifiedTransactionMessageIdentities(identities []txverify.VerifiedMessageIdentity) error { + if b == nil { + return fmt.Errorf("verified message identities do not cover block transactions") + } + prepared, err := PrepareVerifiedTransactionMessageIdentities(b.Transactions, identities) + if err != nil { + if len(identities) != len(b.Transactions) { + return fmt.Errorf("verified message identities do not cover block transactions") + } + return err + } + state := b.transactionState() + state.mu.Lock() + defer state.mu.Unlock() + state.messageIdentities = prepared + return nil +} diff --git a/pkg/block/verified_message_identity_test.go b/pkg/block/verified_message_identity_test.go new file mode 100644 index 000000000..c82d5a4d2 --- /dev/null +++ b/pkg/block/verified_message_identity_test.go @@ -0,0 +1,42 @@ +package block + +import ( + "crypto/ed25519" + "encoding/json" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestVerifiedIdentityCacheOwnsStorageAndDoesNotSerializeTrust(t *testing.T) { + tx := identityTestTransaction(1) + key := ed25519.NewKeyFromSeed(make([]byte, 32)) + tx.Message.AccountKeys[0] = solana.PublicKeyFromBytes(key.Public().(ed25519.PublicKey)) + msg, err := tx.Message.MarshalBinary() + require.NoError(t, err) + tx.Signatures[0] = solana.SignatureFromBytes(ed25519.Sign(key, msg)) + blk := &Block{Transactions: []*solana.Transaction{tx}} + ids := make([]txverify.VerifiedMessageIdentity, 1) + errs := make([]error, 1) + var verifier txverify.BatchVerifier + verifier.VerifyWithMessageIdentities(blk.Transactions, errs, ids) + require.NoError(t, errs[0]) + require.NoError(t, blk.CacheVerifiedTransactionMessageIdentities(ids)) + cached := blk.transactionDerivedState.messageIdentities + require.NotNil(t, cached, "adoption must populate the cache before the first admission lookup") + clear(ids) + got, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + require.Same(t, cached, got) + copyBlock := *blk + got, err = copyBlock.PrepareTransactionMessageIdentities() + require.NoError(t, err) + require.Same(t, cached, got) + wire, err := json.Marshal(blk) + require.NoError(t, err) + var decoded Block + require.NoError(t, json.Unmarshal(wire, &decoded)) + require.Nil(t, decoded.transactionDerivedState) +} diff --git a/pkg/blockprod/bank.go b/pkg/blockprod/bank.go index 26784fdf5..8535e86ec 100644 --- a/pkg/blockprod/bank.go +++ b/pkg/blockprod/bank.go @@ -3,6 +3,7 @@ package blockprod import ( "sync" + "github.com/Overclock-Validator/mithril/pkg/arena" "github.com/Overclock-Validator/mithril/pkg/costmodel" "github.com/Overclock-Validator/mithril/pkg/features" "github.com/Overclock-Validator/mithril/pkg/fees" @@ -44,6 +45,10 @@ type WorkingBank struct { // seenMessages is the bank-local AlreadyProcessed status set. The TPU's // signature LRU is only an ingress optimization and is not authoritative. seenMessages map[[32]byte]struct{} + // Execution and commit are serialized by mu. Borrowed accounts never escape + // either phase, so their storage can be reset for the next transaction. + borrowedAccounts *arena.Arena[sealevel.BorrowedAccount] + preparer *replay.TransactionPreparer } type BankConfig struct { @@ -71,7 +76,12 @@ func NewWorkingBank(cfg BankConfig) *WorkingBank { if sink == nil { sink = NopBatchSink{} } + var preparer *replay.TransactionPreparer + if cfg.SlotCtx != nil { + preparer = replay.NewTransactionPreparer(cfg.SlotCtx.Features) + } return &WorkingBank{ + preparer: preparer, slotCtx: cfg.SlotCtx, slot: cfg.Slot, leader: cfg.Leader, @@ -82,6 +92,7 @@ func NewWorkingBank(cfg BankConfig) *WorkingBank { accepting: true, ancestorStatuses: cfg.TransactionStatuses, seenMessages: make(map[[32]byte]struct{}), + borrowedAccounts: arena.New[sealevel.BorrowedAccount](64), } } @@ -173,13 +184,40 @@ func (b *WorkingBank) Forge(wire []byte) (ForgeResult, costmodel.ExceedReason) { return b.ForgeTransaction(tx, len(wire)) } -// ForgeTransaction executes and commits a parsed transaction. +// ForgeTransaction executes and commits a parsed transaction. The caller must +// keep it immutable: accepted transactions are retained for entry publication. func (b *WorkingBank) ForgeTransaction(tx *solana.Transaction, wireSize int) (ForgeResult, costmodel.ExceedReason) { + return b.forgeTransaction(tx, wireSize, nil) +} + +// ForgePreparedTransaction reuses static work from owned, immutable TPU bytes. +// A different bank feature snapshot falls back to the ordinary execution path. +func (b *WorkingBank) ForgePreparedTransaction(tx *solana.Transaction, wireSize int, prepared *replay.PreparedTransaction) (ForgeResult, costmodel.ExceedReason) { + if b.slotCtx == nil || !b.preparer.Matches(prepared, tx, b.slotCtx.Features) { + prepared = nil + } + return b.forgeTransaction(tx, wireSize, prepared) +} + +func (b *WorkingBank) forgeTransaction(tx *solana.Transaction, wireSize int, prepared *replay.PreparedTransaction) (ForgeResult, costmodel.ExceedReason) { if tx == nil { b.RebateSchedule(wireSize) return ForgeDroppedParse, costmodel.ExceedNone } - messageHash, err := replay.TransactionMessageHash(tx) + // Validate the full wire before execution can charge fees or publish + // account changes. Entry append then reuses this measured size and has no + // fallible serialization step after commit, including on the prepared path. + serializedSize, err := serializedTransactionSize(tx) + if err != nil { + b.RebateSchedule(wireSize) + return ForgeDroppedParse, costmodel.ExceedNone + } + var messageHash [32]byte + if prepared != nil { + messageHash = prepared.MessageHash() + } else { + messageHash, err = replay.TransactionMessageHash(tx) + } if err != nil { b.RebateSchedule(wireSize) return ForgeDroppedParse, costmodel.ExceedNone @@ -201,7 +239,11 @@ func (b *WorkingBank) ForgeTransaction(tx *solana.Transaction, wireSize int) (Fo f := features.NewFeaturesDefault() feats = f } - cost, err = costmodel.EstimateTransactionCost(tx, feats) + if prepared != nil { + cost = prepared.Cost() + } else { + cost, err = costmodel.EstimateTransactionCost(tx, feats) + } if err != nil { b.RebateSchedule(wireSize) return ForgeDroppedParse, costmodel.ExceedNone @@ -238,15 +280,17 @@ func (b *WorkingBank) ForgeTransaction(tx *solana.Transaction, wireSize int) (Fo if reason := b.reserveEntryBytesLocked(wireSize); reason != costmodel.ExceedNone { return ForgeDroppedCost, reason } - if err := fees.PayerCanFund(b.slotCtx, tx); err != nil { + if err := b.preparer.PayerCanFund(b.slotCtx, tx, prepared); err != nil { return ForgeDroppedExecution, costmodel.ExceedNone } - output := replay.LoadAndExecuteTransaction(replay.LoadAndExecuteTransactionInput{ - SlotCtx: b.slotCtx, - Transaction: tx, - LeanResult: true, - }) + output := b.preparer.LoadAndExecute(replay.LoadAndExecuteTransactionInput{ + SlotCtx: b.slotCtx, + Transaction: tx, + LeanResult: true, + SkipTimingMetrics: true, + Arena: b.borrowedAccounts, + }, prepared) if output.ProcessingResult.TransactionError != nil { feeInfo, err := replay.ApplyFeesOnlyTransaction(b.slotCtx, tx, output) if err != nil { @@ -271,7 +315,7 @@ func (b *WorkingBank) ForgeTransaction(tx *solana.Transaction, wireSize int) (Fo b.costs.Record(cost) execCU, loadedCost := actualExecutionUsage(output) b.costs.Rebate(cost, execCU, loadedCost) - if flushed, batchBytes, didFlush := b.entries.Append(*tx, wireSize); didFlush { + if flushed, batchBytes, didFlush := b.entries.appendSerialized(*tx, wireSize, serializedSize); didFlush { b.entryHash = b.entries.CurrentEntryHash() b.sink.OnEntryBatch(flushed, batchBytes) } diff --git a/pkg/blockprod/bank_test.go b/pkg/blockprod/bank_test.go index 53db0d5d4..13e13029f 100644 --- a/pkg/blockprod/bank_test.go +++ b/pkg/blockprod/bank_test.go @@ -797,6 +797,19 @@ func TestEntryBuilderFlush(t *testing.T) { assert.Greater(t, batchBytes, 0) } +func TestEntryBuilderDefaultTargetCoalescesTransactions(t *testing.T) { + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{0xcd}) + for seq := uint64(0); seq < 2; seq++ { + wire := txfixture.MustSignedTransferWire(seq) + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + entries, _, flushed := builder.Append(*tx, len(wire)) + assert.False(t, flushed) + assert.Empty(t, entries) + } + assert.Equal(t, 2, builder.PendingCount()) +} + func TestControllerWorkingBank(t *testing.T) { controller := NewController() assert.Nil(t, controller.WorkingBank()) @@ -806,3 +819,74 @@ func TestControllerWorkingBank(t *testing.T) { controller.SetWorkingBank(env.Bank) assert.Equal(t, env.Bank, controller.WorkingBank()) } + +func TestWorkingBankSerializationFailureDoesNotCommit(t *testing.T) { + for _, mode := range []string{"ordinary", "prepared"} { + t.Run(mode, func(t *testing.T) { + sink := &captureSink{} + env := NewTestEnv(TestEnvConfig{Sink: sink}) + defer env.Close() + env.SlotCtx.Features.EnableFeature(features.EnableTxV1, 0) + env.Bank.preparer = replay.NewTransactionPreparer(env.SlotCtx.Features) + tx := mustSignedTransfer(t, 1) + _, err := tx.Message.SetVersion(solana.MessageVersionV1) + require.NoError(t, err) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + prepared := env.Bank.preparer.Prepare(tx) + if mode == "prepared" { + require.NotNil(t, prepared) + } + // Inject a malformed signature list after static preparation to + // ensure that even the prepared path cannot skip full-wire validation. + // The message is unchanged and serializable, but the full V1 wire + // cannot encode more signatures than the message header declares. + tx.Signatures = append(tx.Signatures, solana.Signature{1}) + _, err = replay.TransactionMessageHash(tx) + require.NoError(t, err) + _, err = tx.MarshalBinary() + require.ErrorContains(t, err, "signatures but header requires") + payer, err := env.SlotCtx.GetAccount(txfixture.PayerPubkey()) + require.NoError(t, err) + dest, err := env.SlotCtx.GetAccount(txfixture.DestPubkey()) + require.NoError(t, err) + payerBalance, destBalance := payer.Lamports, dest.Lamports + hash := env.Bank.EntryHash() + require.Equal(t, costmodel.ExceedNone, env.Bank.PrepareSchedule(len(wire))) + require.Positive(t, env.Bank.EntryBuilder().ReservedBytes()) + var result ForgeResult + var reason costmodel.ExceedReason + if mode == "prepared" { + result, reason = env.Bank.ForgePreparedTransaction(tx, len(wire), prepared) + } else { + result, reason = env.Bank.ForgeTransaction(tx, len(wire)) + } + require.Equal(t, ForgeDroppedParse, result) + require.Equal(t, costmodel.ExceedNone, reason) + payer, err = env.SlotCtx.GetAccount(txfixture.PayerPubkey()) + require.NoError(t, err) + dest, err = env.SlotCtx.GetAccount(txfixture.DestPubkey()) + require.NoError(t, err) + require.Equal(t, payerBalance, payer.Lamports) + require.Equal(t, destBalance, dest.Lamports) + require.Zero(t, env.Bank.TxFeeAccumulator().TotalFees) + require.Zero(t, env.Bank.CostTracker().BlockCost()) + require.Zero(t, env.Bank.NumSignatures()) + require.Empty(t, env.Bank.ForgedTransactions()) + require.Empty(t, env.Bank.seenMessages) + require.Empty(t, env.SlotCtx.ModifiedAccts) + require.Zero(t, env.Bank.EntryBuilder().PendingCount()) + require.Zero(t, env.Bank.EntryBuilder().ReservedBytes()) + require.Zero(t, env.Bank.EntryBytes()) + require.Equal(t, hash, env.Bank.EntryHash()) + require.Empty(t, sink.batches) + // The same message remains eligible after correcting the wire. + tx.Signatures = tx.Signatures[:1] + result, reason = env.Bank.ForgeTransaction(tx, len(wire)) + require.Equal(t, ForgeAccepted, result) + require.Equal(t, costmodel.ExceedNone, reason) + require.Zero(t, env.Bank.EntryBuilder().ReservedBytes()) + require.Equal(t, 1, env.Bank.EntryBuilder().PendingCount()) + }) + } +} diff --git a/pkg/blockprod/entry.go b/pkg/blockprod/entry.go index 098e3170e..75ad95a19 100644 --- a/pkg/blockprod/entry.go +++ b/pkg/blockprod/entry.go @@ -1,11 +1,10 @@ package blockprod import ( - "bytes" - "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/Overclock-Validator/mithril/pkg/statsd" "github.com/Overclock-Validator/mithril/pkg/turbine" - bin "github.com/gagliardetto/binary" "github.com/gagliardetto/solana-go" ) @@ -17,11 +16,12 @@ const entryBatchOverheadBytes = 8 + 8 + 32 + 8 type EntryBuilder struct { limits costmodel.Limits - pendingTxns []solana.Transaction - pendingWire int - flushedBytes int - reservedBytes int - entryHash solana.Hash + pendingTxns []solana.Transaction + pendingSerializedBytes int + pendingWire int + flushedBytes int + reservedBytes int + entryHash solana.Hash } func NewEntryBuilder(limits costmodel.Limits, entryHash solana.Hash) *EntryBuilder { @@ -102,32 +102,53 @@ func (b *EntryBuilder) dropReservation() { } // Append adds a forged transaction. The pending entry is held until the next -// transaction would overflow one FEC set. A short leftover is only emitted by -// Flush (slot end / Freeze). +// transaction would overflow the configured batch target. A short leftover is only emitted by +// Flush (slot end / Freeze). Appended transactions must remain immutable. func (b *EntryBuilder) Append(tx solana.Transaction, wireSize int) ([]turbine.Entry, int, bool) { + serializedSize, err := serializedTransactionSize(&tx) + if err != nil { + return nil, 0, false + } + return b.appendSerialized(tx, wireSize, serializedSize) +} + +// serializedTransactionSize validates the complete transaction, not just its +// message. In particular, v1 signature-count errors surface only here. +func serializedTransactionSize(tx *solana.Transaction) (int, error) { + wire, err := tx.MarshalBinary() + if err != nil { + _ = statsd.Count(statsd.BlockProductionEntrySerializationErrors, 1, nil) + mlog.Log.Errorf("entry builder: cannot serialize transaction: %v", err) + return 0, err + } + return len(wire), nil +} + +// appendSerialized cannot fail after bank state is applied: the caller has +// already serialized this exact transaction and must keep it immutable. Keep +// canonical component size separate from the transport reservation-size hint. +func (b *EntryBuilder) appendSerialized(tx solana.Transaction, wireSize, serializedSize int) ([]turbine.Entry, int, bool) { if wireSize <= 0 { - wire, err := tx.MarshalBinary() - if err != nil { - return nil, 0, false - } - wireSize = len(wire) + wireSize = serializedSize } b.consumeReserved(wireSize) - if b.wouldOverflowBatch(wireSize) { + if b.wouldOverflowBatch(serializedSize) { flushed, batchBytes := b.flushLocked() b.pendingTxns = append(b.pendingTxns[:0], tx) + b.pendingSerializedBytes = serializedSize b.pendingWire = wireSize return flushed, batchBytes, true } b.pendingTxns = append(b.pendingTxns, tx) + b.pendingSerializedBytes += serializedSize b.pendingWire += wireSize return nil, 0, false } func (b *EntryBuilder) projectedBytes(nextWire int) int { - return entryBatchOverheadBytes + b.pendingWire + nextWire + return entryBatchOverheadBytes + b.pendingSerializedBytes + nextWire } // Flush emits the current pending transactions as a single PoH entry. @@ -151,37 +172,10 @@ func (b *EntryBuilder) flushLocked() ([]turbine.Entry, int) { Txns: txns, }} b.entryHash = entryHash - batchBytes, err := marshalEntryBatchBytes(entries) - if err != nil { - return nil, 0 - } - b.flushedBytes += len(batchBytes) + batchBytes := entryBatchOverheadBytes + b.pendingSerializedBytes + b.flushedBytes += batchBytes b.pendingTxns = b.pendingTxns[:0] b.pendingWire = 0 - return entries, len(batchBytes) -} - -func marshalEntryBatchBytes(entries []turbine.Entry) ([]byte, error) { - var buf bytes.Buffer - enc := bin.NewEncoderWithEncoding(&buf, bin.EncodingBin) - if err := enc.WriteUint64(uint64(len(entries)), bin.LE); err != nil { - return nil, err - } - for _, entry := range entries { - if err := enc.WriteUint64(entry.NumHashes, bin.LE); err != nil { - return nil, err - } - if err := enc.WriteBytes(entry.Hash[:], false); err != nil { - return nil, err - } - if err := enc.WriteUint64(uint64(len(entry.Txns)), bin.LE); err != nil { - return nil, err - } - for i := range entry.Txns { - if err := entry.Txns[i].MarshalWithEncoder(enc); err != nil { - return nil, err - } - } - } - return buf.Bytes(), nil + b.pendingSerializedBytes = 0 + return entries, batchBytes } diff --git a/pkg/blockprod/entry_bench_test.go b/pkg/blockprod/entry_bench_test.go new file mode 100644 index 000000000..08f9bc5e7 --- /dev/null +++ b/pkg/blockprod/entry_bench_test.go @@ -0,0 +1,75 @@ +package blockprod + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" +) + +// Measure the producer's batch accounting both alone and through component +// serialization, shred generation, and broadcast-session bookkeeping. +// Transaction execution, peer routing, and UDP are excluded. +func BenchmarkEntryBuilder(b *testing.B) { + wires := txfixture.PrecomputeTransferPool(512) + txns := make([]solana.Transaction, len(wires)) + for i, wire := range wires { + tx, err := solana.TransactionFromBytes(wire) + if err != nil { + b.Fatal(err) + } + txns[i] = *tx + } + leader := txfixture.PayerPrivateKey() + for _, tc := range []struct { + name string + shred bool + broadcast bool + }{ + {name: "append"}, + {name: "append-and-shred", shred: true}, + {name: "append-and-broadcast", broadcast: true}, + } { + b.Run(tc.name, func(b *testing.B) { + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{}) + shredder := turbine.Shredder{Slot: 100, ParentSlot: 99, Version: 1} + session := turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: leader, Slot: 100, ParentSlot: 99, Version: 1, + Broadcaster: &benchmarkPacketBroadcaster{}, + }) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + index := i % len(txns) + entries, _, flushed := builder.Append(txns[index], len(wires[index])) + if flushed && tc.broadcast { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + } + if flushed && tc.shred { + component, err := turbine.NewEntryBatch(entries) + if err != nil { + b.Fatal(err) + } + if _, _, _, err := shredder.MakeMerkleShredsFromComponent( + leader, component, false, solana.Hash{}, 0, 0, + ); err != nil { + b.Fatal(err) + } + } + } + }) + } +} + +type benchmarkPacketBroadcaster struct { + packets int +} + +func (b *benchmarkPacketBroadcaster) Broadcast(packets [][]byte) error { + b.packets += len(packets) + return nil +} diff --git a/pkg/blockprod/entry_test.go b/pkg/blockprod/entry_test.go index 2a0f88fd9..a22ec5e45 100644 --- a/pkg/blockprod/entry_test.go +++ b/pkg/blockprod/entry_test.go @@ -5,6 +5,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/costmodel" "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" "github.com/gagliardetto/solana-go" "github.com/stretchr/testify/require" ) @@ -76,3 +77,102 @@ func mustTransferTx(t *testing.T, seq uint64) *solana.Transaction { require.NoError(t, err) return tx } + +func TestEntryBuilderBatchSizingMatchesComponentEncoding(t *testing.T) { + for _, tc := range []struct { + name string + count int + v0 bool + }{ + {name: "legacy", count: 2}, + {name: "mixed-version", count: 2, v0: true}, + {name: "more-than-128-transactions", count: 129, v0: true}, + } { + t.Run(tc.name, func(t *testing.T) { + txns := make([]solana.Transaction, tc.count) + for i := range txns { + tx, err := solana.TransactionFromBytes(txfixture.MustSignedTransferWire(uint64(i))) + require.NoError(t, err) + if tc.v0 && i%2 != 0 { + tx.Message.SetVersion(solana.MessageVersionV0) + } + txns[i] = *tx + } + component, err := turbine.NewEntryBatch([]turbine.Entry{{NumHashes: 1, Txns: txns}}) + require.NoError(t, err) + encoded, err := turbine.MarshalBlockComponent(component) + require.NoError(t, err) + limits := costmodel.DefaultLimits() + limits.MaxBatchBytes = uint64(len(encoded)) + builder := NewEntryBuilder(limits, solana.Hash{1}) + + totalFlushed := 0 + checkBatch := func(entries []turbine.Entry, batchBytes int, want []solana.Transaction) { + t.Helper() + require.Len(t, entries, 1) + require.Equal(t, want, entries[0].Txns) + component, err := turbine.NewEntryBatch(entries) + require.NoError(t, err) + encoded, err := turbine.MarshalBlockComponent(component) + require.NoError(t, err) + require.Equal(t, len(encoded), batchBytes) + totalFlushed += batchBytes + require.Equal(t, totalFlushed, builder.FlushedBytes()) + } + + // Repeat after both an automatic and explicit flush to catch stale + // size accounting. The limit fits the complete batch exactly. + for round := 0; round < 2; round++ { + pendingWire := 0 + for i := range txns { + wire, err := txns[i].MarshalBinary() + require.NoError(t, err) + wireHint := len(wire) + switch i % 3 { + case 0: + wireHint += 100 // Transport bytes must not affect batch sizing. + case 1: + wireHint = 0 // An absent hint must use the serialized length. + } + if wireHint > 0 { + pendingWire += wireHint + } else { + pendingWire += len(wire) + } + entries, _, flushed := builder.Append(txns[i], wireHint) + require.False(t, flushed, "transaction %d", i) + require.Empty(t, entries) + } + require.Equal(t, tc.count, builder.PendingCount()) + require.Equal(t, pendingWire, builder.PendingWireBytes()) + require.Equal(t, len(encoded), builder.projectedBytes(0)) + entries, batchBytes, flushed := builder.Append(txns[0], 0) + require.True(t, flushed) + checkBatch(entries, batchBytes, txns) + require.Equal(t, int(limits.MaxBatchBytes), batchBytes) + require.Equal(t, 1, builder.PendingCount()) + entries, batchBytes = builder.Flush() + checkBatch(entries, batchBytes, txns[:1]) + require.Zero(t, builder.PendingCount()) + require.Zero(t, builder.PendingWireBytes()) + } + }) + } +} + +func TestEntryBuilderAllowsSingleTransactionOverBatchTarget(t *testing.T) { + wire := txfixture.MustSignedTransferWire(0) + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + limits := costmodel.DefaultLimits() + limits.MaxBatchBytes = 1 + builder := NewEntryBuilder(limits, solana.Hash{}) + entries, _, flushed := builder.Append(*tx, len(wire)) + require.False(t, flushed) + require.Empty(t, entries) + entries, _, flushed = builder.Append(*tx, len(wire)) + require.True(t, flushed) + require.Len(t, entries, 1) + require.Len(t, entries[0].Txns, 1) + require.Equal(t, 1, builder.PendingCount()) +} diff --git a/pkg/blockprod/leader.go b/pkg/blockprod/leader.go index e304ffb8d..7dc150967 100644 --- a/pkg/blockprod/leader.go +++ b/pkg/blockprod/leader.go @@ -97,17 +97,19 @@ type LeaderLoop struct { alpenglowClock bool parentContext func(uint64) ParentContext productionParent func(uint64) alpenglow.BlockProductionParent + canSignSlot func(uint64) bool onBlock func(*b.Block) commitLeaderSlot func(replay.CommitLeaderInput) (*sealevel.SlotCtx, error) currentSlot func() uint64 leaderForSlot func(uint64) (solana.PublicKey, bool) - pollInterval time.Duration - slotDuration time.Duration - now func() time.Time - tickStartedAt time.Time - tickDeliveryLag time.Duration + pollInterval time.Duration + slotDuration time.Duration + completionReserve time.Duration + now func() time.Time + tickStartedAt time.Time + tickDeliveryLag time.Duration mu sync.Mutex activeSlot uint64 @@ -145,6 +147,7 @@ type LeaderLoopConfig struct { AlpenglowClock bool ParentContext func(uint64) ParentContext ProductionParent func(slot uint64) alpenglow.BlockProductionParent + CanSignSlot func(slot uint64) bool OnBlock func(*b.Block) CurrentSlot func() uint64 LeaderForSlot func(uint64) (solana.PublicKey, bool) @@ -154,7 +157,10 @@ type LeaderLoopConfig struct { RewardCerts RewardCertBuilder PollInterval time.Duration SlotDuration time.Duration - Now func() time.Time + // CompletionReserve is time retained for finalization and broadcast. Zero + // uses the conservative default; tune only from measured completion times. + CompletionReserve time.Duration + Now func() time.Time } func NewLeaderLoop(cfg LeaderLoopConfig) *LeaderLoop { @@ -170,6 +176,9 @@ func NewLeaderLoop(cfg LeaderLoopConfig) *LeaderLoop { if cfg.Now == nil { cfg.Now = time.Now } + if cfg.CompletionReserve <= 0 { + cfg.CompletionReserve = leaderBlockCompletionReserve + } return &LeaderLoop{ controller: cfg.Controller, identity: cfg.Identity, @@ -181,12 +190,14 @@ func NewLeaderLoop(cfg LeaderLoopConfig) *LeaderLoop { alpenglowClock: cfg.AlpenglowClock, parentContext: cfg.ParentContext, productionParent: cfg.ProductionParent, + canSignSlot: cfg.CanSignSlot, onBlock: cfg.OnBlock, currentSlot: cfg.CurrentSlot, leaderForSlot: cfg.LeaderForSlot, rewardCerts: cfg.RewardCerts, pollInterval: cfg.PollInterval, slotDuration: cfg.SlotDuration, + completionReserve: cfg.CompletionReserve, now: cfg.Now, finishedLeaderSlots: make(map[uint64]struct{}), pendingFailures: make(map[uint64]leaderSlotFailure), @@ -329,9 +340,11 @@ func (l *LeaderLoop) tickScheduled(scheduledAt time.Time) { delete(l.pendingFailures, targetSlot) openedAt := l.now() l.recordTickTimingLocked(openedAt, "opened") - mlog.Log.InfofPrecise("ALPENGLOW block production: opened local leader slot=%d parent_slot=%d replay_frontier=%d live_slot=%d start_slot_ms=%d%s", + limits := l.activeBank.CostTracker().Limits() + mlog.Log.InfofPrecise("ALPENGLOW block production: opened local leader slot=%d parent_slot=%d replay_frontier=%d live_slot=%d start_slot_ms=%d block_cost_limit=%d account_cost_limit=%d entry_bytes_limit=%d%s", targetSlot, l.parentCtx.ParentSlot, global.ReplayFrontier(), wallSlot, - startDuration.Milliseconds(), l.productionStartTimingDetailLocked(targetSlot, openedAt)) + startDuration.Milliseconds(), limits.BlockCost, limits.WritableAccountCost, limits.MaxEntryBytes, + l.productionStartTimingDetailLocked(targetSlot, openedAt)) return } @@ -586,7 +599,10 @@ func (l *LeaderLoop) productionWindowDeadlineLocked(slot uint64) time.Time { if deadline.IsZero() { return time.Time{} } - reserve := leaderBlockCompletionReserve + reserve := l.completionReserve + if reserve <= 0 { + reserve = leaderBlockCompletionReserve + } if reserve >= l.slotDuration { reserve = l.slotDuration / 4 } @@ -1101,6 +1117,9 @@ func (l *LeaderLoop) revalidateProductionParentForStartLocked(slot uint64, selec } func (l *LeaderLoop) startSlotLocked(slot uint64) error { + if l.canSignSlot != nil && !l.canSignSlot(slot) { + return fmt.Errorf("%w: waiting for durable signing reservation", errParentNotReady) + } selectedParent, parentReadyRequired, err := l.resolveProductionParent(slot) if err != nil { return err @@ -1205,12 +1224,16 @@ func (l *LeaderLoop) startSlotLocked(slot uint64) error { if startEntryHash == (solana.Hash{}) { startEntryHash = parentCtx.ParentBankhash } + limits, err := costmodel.LimitsForSlot(slotCtx.Features, epochSchedule, slot) + if err != nil { + return fmt.Errorf("leader slot limits: %w", err) + } sink := NewShredSink(session) bank := NewWorkingBank(BankConfig{ SlotCtx: slotCtx, Slot: slot, Leader: l.identity.PublicKey(), - Limits: costmodel.LimitsForFeatures(slotCtx.Features), + Limits: limits, EntryHash: startEntryHash, Sink: sink, TransactionStatuses: parentCtx.TransactionStatuses, diff --git a/pkg/blockprod/leader_completion_reserve_test.go b/pkg/blockprod/leader_completion_reserve_test.go new file mode 100644 index 000000000..062bf84d4 --- /dev/null +++ b/pkg/blockprod/leader_completion_reserve_test.go @@ -0,0 +1,32 @@ +package blockprod + +import ( + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +func TestConfiguredCompletionReserveKeepsProtocolDeadlines(t *testing.T) { + ready := time.Unix(1_700_000_000, 0) + now := ready + loop := NewLeaderLoop(LeaderLoopConfig{ + SlotDuration: AlpenglowSlotDuration, CompletionReserve: 60 * time.Millisecond, + Now: func() time.Time { return now }, + }) + loop.productionWindow = leaderProductionWindow{ + active: true, startSlot: 212, endSlot: 215, nextSlot: 212, readyAt: ready, + } + for offset := uint64(0); offset < 4; offset++ { + slot := 212 + offset + deadline := ready.Add(time.Duration(offset+1) * 200 * time.Millisecond) + require.Equal(t, deadline, loop.productionWindowProtocolDeadlineLocked(slot)) + require.Equal(t, deadline.Add(-60*time.Millisecond), loop.productionWindowDeadlineLocked(slot)) + // The override admits starts during the additional 15ms, but still + // rejects at its own cutoff instead of extending the protocol deadline. + now = deadline.Add(-70 * time.Millisecond) + require.NoError(t, loop.productionStartCutoffErrorLocked(slot)) + now = deadline.Add(-60 * time.Millisecond) + require.ErrorIs(t, loop.productionStartCutoffErrorLocked(slot), errProductionStartCutoffElapsed) + } +} diff --git a/pkg/blockprod/leader_processing_test.go b/pkg/blockprod/leader_processing_test.go new file mode 100644 index 000000000..71303a2cf --- /dev/null +++ b/pkg/blockprod/leader_processing_test.go @@ -0,0 +1,81 @@ +package blockprod + +import ( + "sync/atomic" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/arena" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/replay" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestLeaderExecutionOptionsPreserveResults(t *testing.T) { + for _, testCase := range []string{"success", "instruction_failure", "payer_failure"} { + t.Run(testCase, func(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + tx, err := solana.TransactionFromBytes(txfixture.MustSignedTransferWire(42)) + require.NoError(t, err) + if testCase == "instruction_failure" { + tx.Message.Instructions[0].Data[0] = 0xff + } + if testCase == "payer_failure" { + setPayerLamports(t, env, 1) + } + input := replay.LoadAndExecuteTransactionInput{SlotCtx: env.SlotCtx, Transaction: tx, LeanResult: true} + reference := replay.LoadAndExecuteTransaction(input) + input.SkipTimingMetrics = true + // A one-object arena also exercises the heap fallback. Repeated calls + // reset it, while the previous result's account state remains valid. + input.Arena = arena.New[sealevel.BorrowedAccount](1) + before := atomic.LoadUint64(&metrics.GlobalBlockReplay.InstructionsAndAccountMetasFromTx.Count) + beforeDispatch := atomic.LoadUint64(&metrics.GlobalBlockReplay.GetNextIxCtx.Count) + for i := 0; i < 3; i++ { + got := replay.LoadAndExecuteTransaction(input) + require.Equal(t, reference.ProcessingResult, got.ProcessingResult) + require.Equal(t, reference.FeeInfo, got.FeeInfo) + require.Equal(t, reference.LoadedAccountsDataSize, got.LoadedAccountsDataSize) + if reference.ExecCtx != nil { + require.NotNil(t, got.ExecCtx) + require.Equal(t, reference.ExecCtx.ComputeMeter.Used(), got.ExecCtx.ComputeMeter.Used()) + require.Equal(t, reference.ExecCtx.TransactionContext.Accounts.Accounts, got.ExecCtx.TransactionContext.Accounts.Accounts) + } + } + require.Equal(t, before, atomic.LoadUint64(&metrics.GlobalBlockReplay.InstructionsAndAccountMetasFromTx.Count)) + require.Equal(t, beforeDispatch, atomic.LoadUint64(&metrics.GlobalBlockReplay.GetNextIxCtx.Count)) + }) + } +} + +func TestWorkingBankReusesBorrowedAccountsWithoutChangingMessages(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + payerBefore, err := env.SlotCtx.GetAccount(txfixture.PayerPubkey()) + require.NoError(t, err) + payerBalance := payerBefore.Lamports + destBefore, err := env.SlotCtx.GetAccount(txfixture.DestPubkey()) + require.NoError(t, err) + destBalance := destBefore.Lamports + const count = 130 + for i := 0; i < count; i++ { + wire := txfixture.MustSignedTransferWire(uint64(i)) + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + result, _ := env.Bank.ForgeTransaction(tx, len(wire)) + require.Equal(t, ForgeAccepted, result) + after, err := tx.MarshalBinary() + require.NoError(t, err) + require.Equal(t, wire, after) + } + payerAfter, err := env.SlotCtx.GetAccount(txfixture.PayerPubkey()) + require.NoError(t, err) + destAfter, err := env.SlotCtx.GetAccount(txfixture.DestPubkey()) + require.NoError(t, err) + const transferred = count * (count + 1) / 2 + require.Equal(t, payerBalance-transferred-count*5000, payerAfter.Lamports) + require.Equal(t, destBalance+transferred, destAfter.Lamports) +} diff --git a/pkg/blockprod/leader_signing_reservation_test.go b/pkg/blockprod/leader_signing_reservation_test.go new file mode 100644 index 000000000..4b7e09c10 --- /dev/null +++ b/pkg/blockprod/leader_signing_reservation_test.go @@ -0,0 +1,17 @@ +package blockprod + +import ( + "github.com/stretchr/testify/require" + "testing" +) + +func TestSigningReservationGatesEveryLeaderSlotBeforeBuild(t *testing.T) { + for slot := uint64(40); slot < 44; slot++ { + var checked uint64 + l := &LeaderLoop{canSignSlot: func(s uint64) bool { checked = s; return false }} + require.ErrorIs(t, l.startSlotLocked(slot), errParentNotReady) + require.Equal(t, slot, checked) + // All other builder dependencies are deliberately nil: rejection must happen + // before accessing a working bank, executing or signing any shreds. + } +} diff --git a/pkg/blockprod/leader_throughput_bench_test.go b/pkg/blockprod/leader_throughput_bench_test.go new file mode 100644 index 000000000..006633c93 --- /dev/null +++ b/pkg/blockprod/leader_throughput_bench_test.go @@ -0,0 +1,117 @@ +package blockprod + +import ( + "math" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/replay" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" + computebudget "github.com/gagliardetto/solana-go/programs/compute-budget" +) + +// BenchmarkWorkingBankHotAccounts measures admission, execution, publication, +// and entry batching. Signing and bank setup are outside the measured region. +// All wires are unique within a bank, so this cannot benchmark dedup rejection. +func BenchmarkWorkingBankHotAccounts(b *testing.B) { benchmarkWorkingBankHotAccounts(b, "wire") } +func BenchmarkWorkingBankDecodedHotAccounts(b *testing.B) { + benchmarkWorkingBankHotAccounts(b, "decoded") +} +func BenchmarkWorkingBankPreparedHotAccounts(b *testing.B) { + benchmarkWorkingBankHotAccounts(b, "prepared") +} + +func benchmarkWorkingBankHotAccounts(b *testing.B, mode string) { + const perBank = 10000 + for _, workload := range []string{"compute_budget", "transfer"} { + b.Run(workload, func(b *testing.B) { + wires := make([][]byte, perBank) + for i := range wires { + if workload == "transfer" { + wires[i] = txfixture.MustSignedTransferWire(uint64(i)) + continue + } + tx, err := solana.NewTransaction([]solana.Instruction{ + computebudget.NewSetComputeUnitLimitInstruction(uint32(1000 + i)).Build(), + }, txfixture.TestBlockhash(), solana.TransactionPayer(txfixture.PayerPubkey())) + if err != nil { + b.Fatal(err) + } + key := txfixture.PayerPrivateKey() + if _, err = tx.Sign(func(solana.PublicKey) *solana.PrivateKey { return &key }); err != nil { + b.Fatal(err) + } + wires[i], err = tx.MarshalBinary() + if err != nil { + b.Fatal(err) + } + } + decoded := make([]*solana.Transaction, perBank) + for i, wire := range wires { + var err error + decoded[i], err = solana.TransactionFromBytes(wire) + if err != nil { + b.Fatal(err) + } + } + var prepared []*replay.PreparedTransaction + var env *TestEnv + defer func() { + if env != nil { + env.Close() + } + }() + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if i%perBank == 0 { + b.StopTimer() + if env != nil { + env.Close() + } + env = NewTestEnv(TestEnvConfig{}) + env.SlotCtx.Features.EnableFeature(features.RemoveAccountsDeltaHash, 0) + env.SlotCtx.Features.EnableFeature(features.RaiseBlockLimitsTo100m, 0) + if err := env.SlotCtx.Accounts.SetAccount(&addresses.ComputeBudgetProgramAddr, &accounts.Account{ + Key: addresses.ComputeBudgetProgramAddr, Lamports: 1, + Owner: addresses.NativeLoaderAddr, Executable: true, RentEpoch: math.MaxUint64, + }); err != nil { + b.Fatal(err) + } + + if mode == "prepared" { + // Bind the immutable snapshot after all fixture feature setup is complete. + env.Bank.preparer = replay.NewTransactionPreparer(env.SlotCtx.Features) + if prepared == nil { + prepared = make([]*replay.PreparedTransaction, perBank) + for j, tx := range decoded { + prepared[j] = env.Bank.preparer.Prepare(tx) + if prepared[j] == nil { + b.Fatal("preparation failed") + } + } + } + } + b.StartTimer() + } + var result ForgeResult + var reason costmodel.ExceedReason + switch mode { + case "prepared": + result, reason = env.Bank.ForgePreparedTransaction(decoded[i%perBank], len(wires[i%perBank]), prepared[i%perBank]) + case "decoded": + result, reason = env.Bank.ForgeTransaction(decoded[i%perBank], len(wires[i%perBank])) + default: + result, reason = env.Bank.Forge(wires[i%perBank]) + } + if result != ForgeAccepted { + b.Fatalf("transaction %d: %v / %v", i, result, reason) + } + } + }) + } +} diff --git a/pkg/blockprod/prepared_transaction_test.go b/pkg/blockprod/prepared_transaction_test.go new file mode 100644 index 000000000..9a6153d64 --- /dev/null +++ b/pkg/blockprod/prepared_transaction_test.go @@ -0,0 +1,142 @@ +package blockprod + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/replay" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" + computebudget "github.com/gagliardetto/solana-go/programs/compute-budget" + "github.com/gagliardetto/solana-go/programs/system" + "github.com/stretchr/testify/require" +) + +func TestPreparedBankPreservesOutcomes(t *testing.T) { + for _, kind := range []string{"success", "instruction_failure", "fees_only", "payer_changed", "expired", "duplicate", "foreign_message"} { + t.Run(kind, func(t *testing.T) { + reference := NewTestEnv(TestEnvConfig{}) + defer reference.Close() + candidate := NewTestEnv(TestEnvConfig{}) + defer candidate.Close() + tx := mustSignedTransfer(t, 7) + if kind == "instruction_failure" { + tx.Message.Instructions[0].Data[0] = 0xff + } + if kind == "fees_only" { + tx = mustSignBankTestTransaction(t, + computebudget.NewSetLoadedAccountsDataSizeLimitInstruction(1).Build(), + system.NewTransferInstruction(1, txfixture.PayerPubkey(), txfixture.DestPubkey()).Build()) + } + if kind == "expired" { + tx.Message.RecentBlockhash = solana.Hash{99} + } + prepared := replay.NewTransactionPreparer(candidate.SlotCtx.Features.Clone()).Prepare(tx) + require.NotNil(t, prepared) + estimate, err := costmodel.EstimateTransactionCost(tx, candidate.SlotCtx.Features) + require.NoError(t, err) + require.Equal(t, estimate, prepared.Cost()) + if kind == "payer_changed" { + // Preparation succeeded while the payer was funded. Admission must + // still reject after another transaction spends that balance. + setPayerLamports(t, reference, 1) + setPayerLamports(t, candidate, 1) + } + if kind == "foreign_message" { + tx = mustSignedTransfer(t, 19) + } + wire, err := tx.MarshalBinary() + require.NoError(t, err) + for i := 0; i < 2; i++ { + a, ar := reference.Bank.ForgeTransaction(tx, len(wire)) + b, br := candidate.Bank.ForgePreparedTransaction(tx, len(wire), prepared) + require.Equal(t, a, b) + require.Equal(t, ar, br) + require.Equal(t, reference.Bank.CostTracker().BlockCost(), candidate.Bank.CostTracker().BlockCost()) + require.Equal(t, reference.Bank.TxFeeAccumulator(), candidate.Bank.TxFeeAccumulator()) + for _, key := range []solana.PublicKey{txfixture.PayerPubkey(), txfixture.DestPubkey()} { + ra, err := reference.SlotCtx.GetAccount(key) + require.NoError(t, err) + ca, err := candidate.SlotCtx.GetAccount(key) + require.NoError(t, err) + require.Equal(t, ra, ca) + } + if kind != "duplicate" { + break + } + } + after, err := tx.MarshalBinary() + require.NoError(t, err) + require.Equal(t, wire, after) + }) + } +} + +func TestPreparedBankRejectsStaleFeatures(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + future := env.SlotCtx.Features.Clone() + future.EnableFeature(features.EnableTxV1, 0) + tx := mustSignedTransfer(t, 1) + _, err := tx.Message.SetVersion(solana.MessageVersionV1) + require.NoError(t, err) + prepared := replay.NewTransactionPreparer(future).Prepare(tx) + require.NotNil(t, prepared) + require.False(t, env.Bank.preparer.Matches(prepared, tx, env.SlotCtx.Features)) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + result, _ := env.Bank.ForgePreparedTransaction(tx, len(wire), prepared) + require.Equal(t, ForgeDroppedExecution, result) + require.Empty(t, env.Bank.ForgedTransactions()) + require.Zero(t, env.Bank.TxFeeAccumulator().TotalFees) +} + +func TestPreparedLeaderStillRejectsFeePayerNoOp(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + // Establish the complete feature snapshot before constructing the bank. + env.SlotCtx.Features.EnableFeature(features.RelaxFeePayerConstraint, 0) + bank := NewWorkingBank(BankConfig{SlotCtx: env.SlotCtx, Slot: env.SlotCtx.Slot, + TransactionStatuses: replay.NewTransactionStatusCache().View()}) + tx := mustSignedTransfer(t, 1) + prepared := bank.preparer.Prepare(tx) + require.NotNil(t, prepared) + setPayerLamports(t, env, 1) + preview := bank.preparer.LoadAndExecute(replay.LoadAndExecuteTransactionInput{ + SlotCtx: env.SlotCtx, Transaction: tx, LeanResult: true}, prepared) + require.True(t, preview.ProcessedAsNoOp) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + result, _ := bank.ForgePreparedTransaction(tx, len(wire), prepared) + require.Equal(t, ForgeDroppedExecution, result) + require.Empty(t, bank.ForgedTransactions()) + require.Zero(t, bank.TxFeeAccumulator().TotalFees) +} + +func TestPreparedExecutionRechecksStateWithoutMutatingPreparation(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + tx := mustSignedTransfer(t, 1) + p := replay.NewTransactionPreparer(env.SlotCtx.Features) + prepared := p.Prepare(tx) + require.NotNil(t, prepared) + input := replay.LoadAndExecuteTransactionInput{SlotCtx: env.SlotCtx, Transaction: tx, LeanResult: true} + for _, balance := range []uint64{10_000_000, 1, 20_000_000} { + setPayerLamports(t, env, balance) + got := p.LoadAndExecute(input, prepared) + want := replay.LoadAndExecuteTransaction(input) + require.Equal(t, want.ProcessingResult, got.ProcessingResult) + require.Equal(t, want.FeeInfo, got.FeeInfo) + require.Equal(t, want.LoadedAccountsDataSize, got.LoadedAccountsDataSize) + if want.ExecCtx != nil { + require.Equal(t, want.ExecCtx.TransactionContext.Accounts.Accounts, got.ExecCtx.TransactionContext.Accounts.Accounts) + require.Equal(t, want.ExecCtx.ComputeMeter.Used(), got.ExecCtx.ComputeMeter.Used()) + } + } + // Rent-boundary checks still use the bank's current payer state. + rent := sealevel.NewDefaultRentSysvar() + setPayerLamports(t, env, rent.MinimumBalance(0)+4999) + require.Error(t, p.PayerCanFund(env.SlotCtx, tx, prepared)) +} diff --git a/pkg/blockprod/producer_block_bench_test.go b/pkg/blockprod/producer_block_bench_test.go new file mode 100644 index 000000000..c93c6849d --- /dev/null +++ b/pkg/blockprod/producer_block_bench_test.go @@ -0,0 +1,66 @@ +package blockprod + +import ( + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "testing" +) + +func BenchmarkProducerBlock50k(b *testing.B) { + const transactions = 50000 + wires := txfixture.PrecomputeTransferPool(512) + txns := make([]solana.Transaction, len(wires)) + for i, wire := range wires { + tx, err := solana.TransactionFromBytes(wire) + if err != nil { + b.Fatal(err) + } + txns[i] = *tx + } + leader := txfixture.PayerPrivateKey() + var lastBatches, lastPackets int + b.ReportAllocs() + b.ResetTimer() + for round := 0; round < b.N; round++ { + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{}) + sink := &benchmarkPacketBroadcaster{} + session := turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: leader, Slot: 100, ParentSlot: 99, Version: 1, Broadcaster: sink, + }) + if err := session.BroadcastHeader(solana.Hash{}); err != nil { + b.Fatal(err) + } + batches := 0 + for i := 0; i < transactions; i++ { + idx := i % len(txns) + entries, _, flushed := builder.Append(txns[idx], len(wires[idx])) + if flushed { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + batches++ + } + } + if entries, _ := builder.Flush(); len(entries) > 0 { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + batches++ + } + if err := session.BroadcastFooter(solana.Hash{1}, 0, nil, nil); err != nil { + b.Fatal(err) + } + if err := session.BroadcastEndingTickLast(builder.CurrentEntryHash()); err != nil { + b.Fatal(err) + } + _ = session.BlockID(99, solana.Hash{}) + lastBatches, lastPackets = batches, sink.packets + } + b.StopTimer() + b.ReportMetric(transactions, "transactions/op") + b.ReportMetric(float64(len(wires[0])), "wire-B/tx") + b.ReportMetric(float64(lastBatches), "entry-batches/op") + b.ReportMetric(float64(lastPackets), "packets/op") +} diff --git a/pkg/blockprod/producer_branch_bench_test.go b/pkg/blockprod/producer_branch_bench_test.go new file mode 100644 index 000000000..93b9d92e6 --- /dev/null +++ b/pkg/blockprod/producer_branch_bench_test.go @@ -0,0 +1,253 @@ +package blockprod + +import ( + "bytes" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + txwire "github.com/Overclock-Validator/mithril/pkg/tpu/wire" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "github.com/gagliardetto/solana-go/programs/system" +) + +// These complete producer CPU workloads compare branch heads using identical +// pre-signed, parsed fixtures and a packet-counting sink. No execution, admission, +// worker queue, routing, UDP, or receiver work is timed. +const branchComparisonOneFECBytes = 32 * 963 +const branchComparisonSlotEntryCap = 20*1024*1024 - 48 + +func branchMaxSizeTransferPool(tb testing.TB) ([][]byte, []solana.Transaction) { + tb.Helper() + payer, destination := txfixture.PayerPubkey(), txfixture.DestPubkey() + privateKey := txfixture.PayerPrivateKey() + wires := make([][]byte, 512) + txns := make([]solana.Transaction, len(wires)) + for i := range wires { + memo := bytes.Repeat([]byte("m"), 981) + copy(memo, fmt.Sprintf("mithril-max-wire-%04d:", i)) + tx, err := solana.NewTransaction([]solana.Instruction{ + system.NewTransferInstruction(uint64(i+1), payer, destination).Build(), + solana.NewInstruction(solana.MemoProgramID, nil, memo), + }, txfixture.TestBlockhash(), solana.TransactionPayer(payer)) + if err != nil { + tb.Fatal(err) + } + _, err = tx.Sign(func(key solana.PublicKey) *solana.PrivateKey { + if key == payer { + return &privateKey + } + return nil + }) + if err != nil { + tb.Fatal(err) + } + raw, err := tx.MarshalBinary() + if err != nil { + tb.Fatal(err) + } + if len(raw) != costmodel.PacketDataSize { + tb.Fatalf("wire size %d, expected %d", len(raw), costmodel.PacketDataSize) + } + if _, err := txwire.Sanitize(raw); err != nil { + tb.Fatal(err) + } + decoded, err := solana.TransactionFromBytes(raw) + if err != nil { + tb.Fatal(err) + } + if err := decoded.VerifySignatures(); err != nil { + tb.Fatal(err) + } + encoded, err := decoded.MarshalBinary() + if err != nil || !bytes.Equal(encoded, raw) { + tb.Fatal("canonical transaction round trip changed bytes") + } + wires[i], txns[i] = raw, *decoded + } + return wires, txns +} + +func branchComparisonFixtures(tb testing.TB, maximum bool) ([][]byte, []solana.Transaction) { + tb.Helper() + if maximum { + return branchMaxSizeTransferPool(tb) + } + wires := txfixture.PrecomputeTransferPool(512) + txns := make([]solana.Transaction, len(wires)) + for i, raw := range wires { + if len(raw) != 215 { + tb.Fatalf("transfer size %d, want 215", len(raw)) + } + if _, err := txwire.Sanitize(raw); err != nil { + tb.Fatal(err) + } + tx, err := solana.TransactionFromBytes(raw) + if err != nil { + tb.Fatal(err) + } + if err := tx.VerifySignatures(); err != nil { + tb.Fatal(err) + } + encoded, err := tx.MarshalBinary() + if err != nil || !bytes.Equal(encoded, raw) { + tb.Fatal("transfer round trip changed bytes") + } + txns[i] = *tx + } + return wires, txns +} + +type branchComparisonSink struct{ packets int } + +func (s *branchComparisonSink) Broadcast(packets [][]byte) error { + s.packets += len(packets) + return nil +} + +type branchComparisonStats struct { + batches, packets, transactions, maxDataShreds, maxEntryBytes int + blockID solana.Hash +} + +func branchComparisonRun(tb testing.TB, wires [][]byte, txns []solana.Transaction, slots []int, limits costmodel.Limits) branchComparisonStats { + tb.Helper() + var stats branchComparisonStats + var parentID, parentRoot solana.Hash + leader := txfixture.PayerPrivateKey() + for number, count := range slots { + slot := uint64(100 + number) + builder := NewEntryBuilder(limits, solana.Hash{}) + sink := &branchComparisonSink{} + session := turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: leader, Slot: slot, ParentSlot: slot - 1, Version: 1, Broadcaster: sink, + ParentBlockID: parentID, ParentChainedMerkleRoot: parentRoot, + }) + if err := session.BroadcastHeader(parentID); err != nil { + tb.Fatal(err) + } + entryBytes := 0 + for i := 0; i < count; i++ { + index := stats.transactions % len(txns) + entries, size, flushed := builder.Append(txns[index], len(wires[index])) + stats.transactions++ + if flushed { + if err := session.BroadcastEntryBatch(entries); err != nil { + tb.Fatal(err) + } + stats.batches++ + entryBytes += size + } + } + if entries, size := builder.Flush(); len(entries) != 0 { + if err := session.BroadcastEntryBatch(entries); err != nil { + tb.Fatal(err) + } + stats.batches++ + entryBytes += size + } + if err := session.BroadcastFooter(solana.Hash{1}, 0, nil, nil); err != nil { + tb.Fatal(err) + } + if err := session.BroadcastEndingTickLast(builder.CurrentEntryHash()); err != nil { + tb.Fatal(err) + } + parentID = session.BlockID(slot-1, parentID) + parentRoot = session.ChainedMerkleRoot() + dataShreds := sink.packets / 2 + if dataShreds > costmodel.DefaultMaxDataShredsPerSlot { + tb.Fatal("slot exceeds data-shred budget") + } + if entryBytes > branchComparisonSlotEntryCap { + tb.Fatal("slot exceeds entry-byte budget") + } + if dataShreds > stats.maxDataShreds { + stats.maxDataShreds = dataShreds + } + if entryBytes > stats.maxEntryBytes { + stats.maxEntryBytes = entryBytes + } + stats.packets += sink.packets + } + stats.blockID = parentID + return stats +} + +func branchComparisonCheck(tb testing.TB, got branchComparisonStats, wireSize int, slots []int, batchLimit uint64) { + tb.Helper() + perBatch := (int(batchLimit) - 56) / wireSize + var batches, packets, transactions int + for _, count := range slots { + full, tail := count/perBatch, count%perBatch + batches += full + fecs := full * ((56 + perBatch*wireSize + branchComparisonOneFECBytes - 1) / branchComparisonOneFECBytes) + if tail != 0 { + batches++ + fecs += (56 + tail*wireSize + branchComparisonOneFECBytes - 1) / branchComparisonOneFECBytes + } + packets += (fecs + 3) * 64 // Header, footer, and signed ending tick each use one FEC set. + transactions += count + } + if got.batches != batches || got.packets != packets || got.transactions != transactions { + tb.Fatalf("workload counts %+v, want batches=%d packets=%d transactions=%d", got, batches, packets, transactions) + } + if got.blockID == (solana.Hash{}) { + tb.Fatal("empty block commitment") + } +} + +func TestProducerBranchComparisonFixtures(t *testing.T) { + for _, maximum := range []bool{false, true} { + wires, txns := branchComparisonFixtures(t, maximum) + limits := costmodel.DefaultLimits() + slots := []int{101, 100} + got := branchComparisonRun(t, wires, txns, slots, limits) + branchComparisonCheck(t, got, len(wires[0]), slots, limits.MaxBatchBytes) + t.Logf("wire=%d batch-target=%d batches=%d packets=%d", len(wires[0]), limits.MaxBatchBytes, got.batches, got.packets) + } +} + +func BenchmarkProducerBranchHeads(b *testing.B) { + for _, workload := range []struct { + name string + maximum bool + slots []int + }{ + {name: "small-215B", slots: []int{50000}}, + {name: "maximum-1232B", maximum: true, slots: []int{16667, 16667, 16666}}, + } { + wires, txns := branchComparisonFixtures(b, workload.maximum) + for _, target := range []struct { + name string + bytes uint64 + }{ + {name: "default"}, + {name: "one-fec", bytes: branchComparisonOneFECBytes}, + } { + b.Run(workload.name+"/"+target.name, func(b *testing.B) { + limits := costmodel.DefaultLimits() + if target.bytes != 0 { + limits.MaxBatchBytes = target.bytes + } + var last branchComparisonStats + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + last = branchComparisonRun(b, wires, txns, workload.slots, limits) + } + b.StopTimer() + branchComparisonCheck(b, last, len(wires[0]), workload.slots, limits.MaxBatchBytes) + b.ReportMetric(float64(last.transactions), "transactions/op") + b.ReportMetric(float64(len(workload.slots)), "slots/op") + b.ReportMetric(float64(len(wires[0])), "wire-B/tx") + b.ReportMetric(float64(limits.MaxBatchBytes), "batch-target-B") + b.ReportMetric(float64(last.batches), "entry-batches/op") + b.ReportMetric(float64(last.packets), "packets/op") + b.ReportMetric(float64(last.maxDataShreds), "max-data-shreds/slot") + b.ReportMetric(float64(last.maxEntryBytes), "max-entry-B/slot") + }) + } + } +} diff --git a/pkg/blockprod/producer_maxsize_bench_test.go b/pkg/blockprod/producer_maxsize_bench_test.go new file mode 100644 index 000000000..e843baf81 --- /dev/null +++ b/pkg/blockprod/producer_maxsize_bench_test.go @@ -0,0 +1,153 @@ +package blockprod + +import ( + "bytes" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + txwire "github.com/Overclock-Validator/mithril/pkg/tpu/wire" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "github.com/gagliardetto/solana-go/programs/system" +) + +// Generate exactly 1,232-byte, single-signature legacy transactions. Padding is +// a real UTF-8 memo instruction, rather than trailing bytes after a transaction. +func maxSizeTransferPool(tb testing.TB) ([][]byte, []solana.Transaction) { + tb.Helper() + payer, destination := txfixture.PayerPubkey(), txfixture.DestPubkey() + privateKey := txfixture.PayerPrivateKey() + wires := make([][]byte, 512) + txns := make([]solana.Transaction, len(wires)) + for i := range wires { + memo := bytes.Repeat([]byte("m"), 981) + copy(memo, fmt.Sprintf("mithril-max-wire-%04d:", i)) + tx, err := solana.NewTransaction([]solana.Instruction{ + system.NewTransferInstruction(uint64(i+1), payer, destination).Build(), + solana.NewInstruction(solana.MemoProgramID, nil, memo), + }, txfixture.TestBlockhash(), solana.TransactionPayer(payer)) + if err != nil { + tb.Fatal(err) + } + _, err = tx.Sign(func(key solana.PublicKey) *solana.PrivateKey { + if key == payer { + return &privateKey + } + return nil + }) + if err != nil { + tb.Fatal(err) + } + raw, err := tx.MarshalBinary() + if err != nil { + tb.Fatal(err) + } + if len(raw) != costmodel.PacketDataSize { + tb.Fatalf("wire size %d, expected %d", len(raw), costmodel.PacketDataSize) + } + if _, err := txwire.Sanitize(raw); err != nil { + tb.Fatal(err) + } + decoded, err := solana.TransactionFromBytes(raw) + if err != nil { + tb.Fatal(err) + } + if err := decoded.VerifySignatures(); err != nil { + tb.Fatal(err) + } + encoded, err := decoded.MarshalBinary() + if err != nil || !bytes.Equal(encoded, raw) { + tb.Fatal("canonical transaction round trip changed bytes") + } + wires[i], txns[i] = raw, *decoded + } + return wires, txns +} + +func TestMaxSizeProducerFixture(t *testing.T) { + wires, txns := maxSizeTransferPool(t) + t.Logf("verified %d signed transactions at %d bytes each", len(wires), len(wires[0])) + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{}) + for i := 0; i < 50; i++ { + entries, batchBytes, flushed := builder.Append(txns[i], len(wires[i])) + if i < 49 && flushed { + t.Fatalf("premature flush at %d", i) + } + if i == 49 { + if !flushed || len(entries) != 1 || len(entries[0].Txns) != 49 || batchBytes != 60424 { + t.Fatal("unexpected full-batch size") + } + } + } +} + +// 50k maximum-size transactions require two slots under the current 32,768 +// data-shred cap. Each iteration completes two slots of 25k transactions. +func BenchmarkProducer50kMaxSize(b *testing.B) { + const transactions = 50000 + const transactionsPerSlot = 25000 + wires, txns := maxSizeTransferPool(b) + leader := txfixture.PayerPrivateKey() + var lastBatches, lastPackets, lastMaxSlotDataShreds int + b.ReportAllocs() + b.ResetTimer() + for round := 0; round < b.N; round++ { + var parentID, parentRoot solana.Hash + totalBatches, totalPackets, maxSlotDataShreds := 0, 0, 0 + for block := 0; block < transactions/transactionsPerSlot; block++ { + slot := uint64(100 + block) + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{}) + sink := &benchmarkPacketBroadcaster{} + session := turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: leader, Slot: slot, ParentSlot: slot - 1, Version: 1, Broadcaster: sink, + ParentBlockID: parentID, ParentChainedMerkleRoot: parentRoot, + }) + if err := session.BroadcastHeader(parentID); err != nil { + b.Fatal(err) + } + for i := 0; i < transactionsPerSlot; i++ { + idx := (block*transactionsPerSlot + i) % len(txns) + entries, _, flushed := builder.Append(txns[idx], len(wires[idx])) + if flushed { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + totalBatches++ + } + } + if entries, _ := builder.Flush(); len(entries) > 0 { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + totalBatches++ + } + if err := session.BroadcastFooter(solana.Hash{1}, 0, nil, nil); err != nil { + b.Fatal(err) + } + if err := session.BroadcastEndingTickLast(builder.CurrentEntryHash()); err != nil { + b.Fatal(err) + } + parentID = session.BlockID(slot-1, parentID) + parentRoot = session.ChainedMerkleRoot() + // The generator emits equal numbers of data and coding shreds. + dataShreds := sink.packets / 2 + if dataShreds > costmodel.DefaultMaxDataShredsPerSlot { + b.Fatal("slot exceeds data-shred limit") + } + if dataShreds > maxSlotDataShreds { + maxSlotDataShreds = dataShreds + } + totalPackets += sink.packets + } + lastBatches, lastPackets, lastMaxSlotDataShreds = totalBatches, totalPackets, maxSlotDataShreds + } + b.StopTimer() + b.ReportMetric(transactions, "transactions/op") + b.ReportMetric(transactions/transactionsPerSlot, "slots/op") + b.ReportMetric(float64(len(wires[0])), "wire-B/tx") + b.ReportMetric(float64(lastBatches), "entry-batches/op") + b.ReportMetric(float64(lastPackets), "packets/op") + b.ReportMetric(float64(lastMaxSlotDataShreds), "max-data-shreds/slot") +} diff --git a/pkg/blockprod/readonly_block_bench_test.go b/pkg/blockprod/readonly_block_bench_test.go new file mode 100644 index 000000000..c605243f5 --- /dev/null +++ b/pkg/blockprod/readonly_block_bench_test.go @@ -0,0 +1,189 @@ +package blockprod + +import ( + "crypto/ed25519" + "crypto/sha256" + "math" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +const readonlyBlockAccepted = 48622 // 50M budget, including the next tx's upfront loaded-data reservation. +const readonlyActualCost = 1028 + +type readonlyBlockParent struct{ mem accounts.MemAccounts } + +func (p readonlyBlockParent) GetAccount(_ uint64, key solana.PublicKey) (*accounts.Account, error) { + raw := [32]byte(key) + return p.mem.GetAccount(&raw) +} + +type readonlyBlockFixture struct { + wires [][]byte + txs []*solana.Transaction + payers []solana.PublicKey + parent readonlyBlockParent +} + +func makeReadonlyBlockFixture(tb testing.TB, count int) readonlyBlockFixture { + tb.Helper() + f := readonlyBlockFixture{parent: readonlyBlockParent{accounts.NewMemAccounts()}} + keys := make([]ed25519.PrivateKey, 8) + for i := range keys { + seed := sha256.Sum256([]byte{byte(i), 73}) + keys[i] = ed25519.NewKeyFromSeed(seed[:]) + f.payers = append(f.payers, solana.PublicKeyFromBytes(keys[i].Public().(ed25519.PublicKey))) + } + pool := make([]solana.PublicKey, txfixture.ReadonlyPairPoolSize) + for i := range pool { + pool[i] = solana.PublicKey{byte(i + 1), 77} + require.NoError(tb, f.parent.mem.SetAccountWithoutLock(pool[i], &accounts.Account{ + Key: pool[i], Lamports: 1_000_000, Data: make([]byte, 9), Owner: solana.PublicKey{11}, RentEpoch: math.MaxUint64, + })) + } + for i := 0; i < count; i++ { + wire, err := txfixture.ReadonlyPairWire(keys[i%8], txfixture.TestBlockhash(), pool, i/8) + require.NoError(tb, err) + tx, err := solana.TransactionFromBytes(wire) + require.NoError(tb, err) + f.wires = append(f.wires, wire) + f.txs = append(f.txs, tx) + } + return f +} + +func (f readonlyBlockFixture) bank(tb testing.TB, sink BatchSink) *TestEnv { + tb.Helper() + // Explicit active 200ms testnet budgets allow identical baseline/candidate + // benchmark fixtures; LimitsForSlot has separate epoch-transition tests. + limits := costmodel.DefaultLimits() + limits.BlockCost, limits.WritableAccountCost = 50_000_000, 20_000_000 + limits.AllocatedDataSizeDelta, limits.MaxEntryBytes = 50_000_000, 10*1024*1024-48 + env := NewTestEnv(TestEnvConfig{Limits: limits, Sink: sink}) + env.SlotCtx.Features.EnableFeature(features.RemoveAccountsDeltaHash, 0) + env.SlotCtx.UnrootedRead = f.parent + for _, payer := range f.payers { + require.NoError(tb, env.SlotCtx.Accounts.SetAccountWithoutLock(payer, &accounts.Account{ + Key: payer, Lamports: 10_000_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: math.MaxUint64, + })) + } + return env +} + +// This is a capacity/correctness test, not a 200ms deadline assertion. It creates +// a full synthetic block, validates fee/cost accounting, and round-trips every +// emitted entry through the production shred generator and decoder. No network. +func TestReadonlyPairBlockCapacityAndShredRoundTrip(t *testing.T) { + f := makeReadonlyBlockFixture(t, readonlyBlockAccepted+1) + sink := &captureSink{} + env := f.bank(t, sink) + defer env.Close() + for i, tx := range f.txs { + result, reason := env.Bank.ForgeTransaction(tx, len(f.wires[i])) + if i < readonlyBlockAccepted { + require.Equal(t, ForgeAccepted, result, "transaction %d", i) + } else { + require.Equal(t, ForgeDroppedCost, result) + require.Equal(t, costmodel.ExceedBlockCost, reason) + } + } + env.Bank.Freeze() + require.Equal(t, uint64(readonlyBlockAccepted*readonlyActualCost), env.Bank.CostTracker().BlockCost()) + require.Equal(t, uint64(readonlyBlockAccepted), env.Bank.NumSignatures()) + require.Equal(t, uint64(readonlyBlockAccepted*5000), env.Bank.TxFeeAccumulator().TotalFees) + require.Len(t, env.Bank.ForgedTransactions(), readonlyBlockAccepted) + require.LessOrEqual(t, uint64(env.Bank.EntryBytes()), env.Bank.CostTracker().Limits().MaxEntryBytes) + for i, payer := range f.payers { + included := readonlyBlockAccepted / 8 + if i < readonlyBlockAccepted%8 { + included++ + } + acct, err := env.SlotCtx.GetAccount(payer) + require.NoError(t, err) + require.Equal(t, uint64(10_000_000_000-included*5000), acct.Lamports) + } + gen := turbine.ShredGenerator{Slot: 42, ParentSlot: 41, Version: 7} + var root solana.Hash + var nextData, nextCode uint32 + count, totalBytes := 0, 0 + previous := solana.Hash{0xab} + for i, entries := range sink.batches { + component, err := turbine.NewEntryBatch(entries) + require.NoError(t, err) + raw, err := turbine.MarshalBlockComponent(component) + require.NoError(t, err) + require.Equal(t, len(raw), sink.bytes[i]) + totalBytes += len(raw) + packets, chained, d, c, err := gen.MakeShredsFromData(txfixture.PayerPrivateKey(), raw, false, root, nextData, nextCode) + require.NoError(t, err) + root, nextData, nextCode = chained, d, c + var shreds []*turbine.Shred + for _, packet := range packets { + sh, err := turbine.ParseShred(packet) + require.NoError(t, err) + if sh.Type == turbine.ShredTypeData { + shreds = append(shreds, sh) + } + } + decoded, err := turbine.DecodeEntriesFromDataShreds(shreds) + require.NoError(t, err) + require.Equal(t, entries, decoded) + for _, entry := range decoded { + require.Equal(t, turbine.NextAlpenglowEntryHash(previous, entry.NumHashes, entry.Txns), entry.Hash) + previous = entry.Hash + for _, tx := range entry.Txns { + require.Equal(t, f.txs[count].Signatures, tx.Signatures) + count++ + } + } + } + require.Equal(t, readonlyBlockAccepted, count) + require.Equal(t, env.Bank.EntryBytes(), totalBytes) + require.Equal(t, env.Bank.EntryHash(), previous) + require.Less(t, nextData, uint32(16384)) // Leaves room for header/footer/ending tick. +} + +// One serial caller; signing and fixture/bank setup are excluded. Measures +// admission, execution, account publication, entry building and final flush. +// It excludes signature verification, actual AccountsDB, network and consensus. +func BenchmarkReadonlyPairFullBlock(b *testing.B) { + f := makeReadonlyBlockFixture(b, readonlyBlockAccepted) + for _, mode := range []string{"wire", "decoded"} { + b.Run(mode, func(b *testing.B) { + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + b.StopTimer() + env := f.bank(b, nil) + b.StartTimer() + for j, wire := range f.wires { + var result ForgeResult + if mode == "wire" { + result, _ = env.Bank.Forge(wire) + } else { + result, _ = env.Bank.ForgeTransaction(f.txs[j], len(wire)) + } + if result != ForgeAccepted { + b.Fatalf("transaction %d: %v", j, result) + } + } + env.Bank.Freeze() + b.StopTimer() + if env.Bank.CostTracker().BlockCost() != readonlyBlockAccepted*readonlyActualCost { + b.Fatal("unexpected block cost") + } + env.Close() + b.StartTimer() + } + b.ReportMetric(float64(readonlyBlockAccepted), "tx/block") + }) + } +} diff --git a/pkg/blockprod/readonly_block_prepared_bench_test.go b/pkg/blockprod/readonly_block_prepared_bench_test.go new file mode 100644 index 000000000..ccf2380de --- /dev/null +++ b/pkg/blockprod/readonly_block_prepared_bench_test.go @@ -0,0 +1,61 @@ +package blockprod + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/replay" +) + +// Models a queue already statically prepared before leadership. Preparation is +// shifted out of this timed bank phase, not eliminated from validator CPU work. +func BenchmarkReadonlyPairPreparedFullBlock(b *testing.B) { + f := makeReadonlyBlockFixture(b, readonlyBlockAccepted) + setup := f.bank(b, nil) + preparer := replay.NewTransactionPreparer(setup.SlotCtx.Features) + prepared := make([]*replay.PreparedTransaction, len(f.txs)) + for i, tx := range f.txs { + prepared[i] = preparer.Prepare(tx) + if prepared[i] == nil { + b.Fatal("preparation failed") + } + } + setup.Close() + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + b.StopTimer() + env := f.bank(b, nil) + env.Bank.preparer = replay.NewTransactionPreparer(env.SlotCtx.Features) + b.StartTimer() + for j, tx := range f.txs { + result, _ := env.Bank.ForgePreparedTransaction(tx, len(f.wires[j]), prepared[j]) + if result != ForgeAccepted { + b.Fatalf("transaction %d: %v", j, result) + } + } + env.Bank.Freeze() + b.StopTimer() + if env.Bank.CostTracker().BlockCost() != readonlyBlockAccepted*readonlyActualCost { + b.Fatal("unexpected block cost") + } + env.Close() + b.StartTimer() + } + b.ReportMetric(float64(readonlyBlockAccepted), "tx/block") +} + +// Allocations per preparation bound the incremental prepared-object footprint; +// excludes the already decoded transaction and wire bytes owned by the queue. +func BenchmarkReadonlyPairPreparationMemory(b *testing.B) { + f := makeReadonlyBlockFixture(b, 1) + setup := f.bank(b, nil) + defer setup.Close() + preparer := replay.NewTransactionPreparer(setup.SlotCtx.Features) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if preparer.Prepare(f.txs[0]) == nil { + b.Fatal("preparation failed") + } + } +} diff --git a/pkg/blockprod/readonly_load_bench_test.go b/pkg/blockprod/readonly_load_bench_test.go new file mode 100644 index 000000000..f5d24798d --- /dev/null +++ b/pkg/blockprod/readonly_load_bench_test.go @@ -0,0 +1,95 @@ +package blockprod + +import ( + "crypto/ed25519" + "math" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/replay" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" +) + +type readonlyBenchParent struct{ mem accounts.MemAccounts } + +func (p readonlyBenchParent) GetAccount(_ uint64, key solana.PublicKey) (*accounts.Account, error) { + raw := [32]byte(key) + return p.mem.GetAccount(&raw) +} + +// Only bank admission/execution/entry building are timed. The immutable parent +// is in memory; this excludes actual AccountsDB, network and signing costs. +func BenchmarkReadonlyPairBank(b *testing.B) { + const perBank = 10000 + parent := readonlyBenchParent{accounts.NewMemAccounts()} + pool := make([]solana.PublicKey, 128) + for i := range pool { + pool[i] = solana.PublicKey{byte(i + 1), 77} + parent.mem.SetAccountWithoutLock(pool[i], &accounts.Account{Key: pool[i], Lamports: 1_000_000, Data: make([]byte, 9), Owner: solana.PublicKey{11}, RentEpoch: math.MaxUint64}) + } + txs := make([]*solana.Transaction, perBank) + key := txfixture.PayerPrivateKey() + payer := key.PublicKey() + hash := txfixture.TestBlockhash() + for i := range txs { + n := (i * 7919) % (128 * 127) + a, c := n/127, n%127 + if c >= a { + c++ + } + msg := []byte{1, 0, 2, 3} + msg = append(msg, payer[:]...) + msg = append(msg, pool[a][:]...) + msg = append(msg, pool[c][:]...) + msg = append(msg, hash[:]...) + msg = append(msg, 0) + wire := []byte{1} + wire = append(wire, ed25519.Sign(ed25519.PrivateKey(key), msg)...) + wire = append(wire, msg...) + var err error + txs[i], err = solana.TransactionFromBytes(wire) + if err != nil { + b.Fatal(err) + } + } + var env *TestEnv + var prepared []*replay.PreparedTransaction + defer func() { + if env != nil { + env.Close() + } + }() + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if i%perBank == 0 { + b.StopTimer() + if env != nil { + env.Close() + } + env = NewTestEnv(TestEnvConfig{}) + env.SlotCtx.Features.EnableFeature(features.RemoveAccountsDeltaHash, 0) + env.SlotCtx.UnrootedRead = parent + env.Bank.preparer = replay.NewTransactionPreparer(env.SlotCtx.Features) + if prepared == nil { + prepared = make([]*replay.PreparedTransaction, perBank) + for j, t := range txs { + prepared[j] = env.Bank.preparer.Prepare(t) + if prepared[j] == nil { + b.Fatal("preparation failed") + } + } + } + b.StartTimer() + } + outcome, reason := env.Bank.ForgePreparedTransaction(txs[i%perBank], 198, prepared[i%perBank]) + if outcome != ForgeAccepted { + b.Fatalf("%v / %v", outcome, reason) + } + if i%perBank == perBank-1 && env.Bank.CostTracker().BlockCost() != perBank*1028 { + b.Fatalf("unexpected cost %d", env.Bank.CostTracker().BlockCost()) + } + } +} diff --git a/pkg/blockprod/scheduler/buffer.go b/pkg/blockprod/scheduler/buffer.go index 307dc0b28..2aa5a70c5 100644 --- a/pkg/blockprod/scheduler/buffer.go +++ b/pkg/blockprod/scheduler/buffer.go @@ -4,15 +4,17 @@ import ( "container/heap" "sync" + "github.com/Overclock-Validator/mithril/pkg/replay" "github.com/gagliardetto/solana-go" ) -// MaxBufferedTxns is the hard cap on cross-slot buffered transactions. +// MaxBufferedTxns is the default cap on cross-slot buffered transactions. const MaxBufferedTxns = 2 * 65536 // entry is one buffered, scored transaction. type entry struct { - tx *solana.Transaction + tx *solana.Transaction + prepared *replay.PreparedTransaction // wire is an owned copy of the packet bytes. Parsed tx fields may alias it // (solana-go decoder slices), so it must outlive any use of tx. wire []byte @@ -25,37 +27,34 @@ type entry struct { // (e.g. cost limit). The entry is retained for cross-slot retry. skipGen uint64 - alive bool - maxIdx int - minIdx int + alive bool + // Indexes belong to Buffer.mu. Every buffered entry appears exactly once + // in each heap; -1 denotes absence while an entry is owned by the consumer. + maxIndex, minIndex int } type maxHeap []*entry func (h maxHeap) Len() int { return len(h) } func (h maxHeap) Less(i, j int) bool { - if h[i].reward != h[j].reward { - return h[i].reward > h[j].reward - } - return h[i].seq < h[j].seq // older first on ties + return higherPriority(h[i], h[j]) } func (h maxHeap) Swap(i, j int) { h[i], h[j] = h[j], h[i] - h[i].maxIdx = i - h[j].maxIdx = j + h[i].maxIndex, h[j].maxIndex = i, j } func (h *maxHeap) Push(x any) { e := x.(*entry) - e.maxIdx = len(*h) + e.maxIndex = len(*h) *h = append(*h, e) } func (h *maxHeap) Pop() any { old := *h n := len(old) e := old[n-1] + e.maxIndex = -1 old[n-1] = nil *h = old[:n-1] - e.maxIdx = -1 return e } @@ -71,21 +70,20 @@ func (h minHeap) Less(i, j int) bool { } func (h minHeap) Swap(i, j int) { h[i], h[j] = h[j], h[i] - h[i].minIdx = i - h[j].minIdx = j + h[i].minIndex, h[j].minIndex = i, j } func (h *minHeap) Push(x any) { e := x.(*entry) - e.minIdx = len(*h) + e.minIndex = len(*h) *h = append(*h, e) } func (h *minHeap) Pop() any { old := *h n := len(old) e := old[n-1] + e.minIndex = -1 old[n-1] = nil *h = old[:n-1] - e.minIdx = -1 return e } @@ -131,27 +129,41 @@ const ( InsertRejectedCapacity ) -// Insert adds e when not a duplicate. At capacity, the lowest-reward entry is -// evicted if e has a strictly higher reward; otherwise e is rejected. -func (b *Buffer) Insert(e *entry) (InsertResult, *entry) { - if e == nil || e.tx == nil { - return InsertRejectedCapacity, nil - } +// precheck rejects entries that cannot be admitted now, without reserving space +// or evicting anything. Preparation runs outside mu; Insert must recheck because +// concurrent arrivals or draining can change both duplicates and the floor. +func (b *Buffer) precheck(e *entry) InsertResult { b.mu.Lock() defer b.mu.Unlock() + return b.admissionLocked(e) +} +func (b *Buffer) admissionLocked(e *entry) InsertResult { + if e == nil || e.tx == nil { + return InsertRejectedCapacity + } if _, exists := b.byHash[e.messageHash]; exists { - return InsertDuplicate, nil + return InsertDuplicate } - var evicted *entry if b.alive >= b.capacity { min := b.peekMinAliveLocked() - if min == nil { - return InsertRejectedCapacity, nil - } - if e.reward <= min.reward { - return InsertRejectedCapacity, nil + if min == nil || e.reward <= min.reward { + return InsertRejectedCapacity } + } + return InsertAccepted +} + +// Insert adds e when not a duplicate. At capacity, the lowest-reward entry is +// evicted if e has a strictly higher reward; otherwise e is rejected. +func (b *Buffer) Insert(e *entry) (InsertResult, *entry) { + b.mu.Lock() + defer b.mu.Unlock() + if result := b.admissionLocked(e); result != InsertAccepted { + return result, nil + } + var evicted *entry + if b.alive >= b.capacity { evicted = b.popMinAliveLocked() } b.pushAliveLocked(e) @@ -173,21 +185,19 @@ func (b *Buffer) Cleanup(drop func(*entry) bool) int { b.mu.Lock() defer b.mu.Unlock() - var doomed []*entry + dropped := 0 for _, e := range b.byHash { - if e.alive && drop(e) { - doomed = append(doomed, e) + if drop(e) { + b.killLocked(e) + dropped++ } } - for _, e := range doomed { - b.killLocked(e) - } - b.drainDeadLocked() - return len(doomed) + return dropped } func (b *Buffer) pushAliveLocked(e *entry) { e.alive = true + e.maxIndex, e.minIndex = -1, -1 b.byHash[e.messageHash] = e heap.Push(&b.max, e) heap.Push(&b.min, e) @@ -198,56 +208,79 @@ func (b *Buffer) killLocked(e *entry) { if e == nil || !e.alive { return } + if e.maxIndex >= 0 { + heap.Remove(&b.max, e.maxIndex) + } + if e.minIndex >= 0 { + heap.Remove(&b.min, e.minIndex) + } e.alive = false delete(b.byHash, e.messageHash) b.alive-- } func (b *Buffer) popMaxAliveLocked() *entry { - for b.max.Len() > 0 { - e := heap.Pop(&b.max).(*entry) - if !e.alive { - continue - } + if b.max.Len() > 0 { + e := b.max.popEntry() b.killLocked(e) return e } return nil } -func (b *Buffer) peekMinAliveLocked() *entry { - for b.min.Len() > 0 { - if b.min[0].alive { - return b.min[0] +// popEntry moves the winning child into the hole at each level. This avoids +// interface dispatch and swapping two entries at every level of a large queue. +// Ordering is identical to maxHeap.Less, including FIFO for equal rewards. +func (h *maxHeap) popEntry() *entry { + nodes := *h + root := nodes[0] + root.maxIndex = -1 + last := nodes[len(nodes)-1] + nodes[len(nodes)-1] = nil + nodes = nodes[:len(nodes)-1] + if len(nodes) > 0 { + i := 0 + for { + child := 2*i + 1 + if child >= len(nodes) { + break + } + if child+1 < len(nodes) && higherPriority(nodes[child+1], nodes[child]) { + child++ + } + if !higherPriority(nodes[child], last) { + break + } + nodes[i] = nodes[child] + nodes[i].maxIndex = i + i = child } - heap.Pop(&b.min) + nodes[i] = last + last.maxIndex = i + } + *h = nodes + return root +} + +func higherPriority(a, b *entry) bool { + if a.reward != b.reward { + return a.reward > b.reward + } + return a.seq < b.seq +} + +func (b *Buffer) peekMinAliveLocked() *entry { + if b.min.Len() > 0 { + return b.min[0] } return nil } func (b *Buffer) popMinAliveLocked() *entry { - for b.min.Len() > 0 { + if b.min.Len() > 0 { e := heap.Pop(&b.min).(*entry) - if !e.alive { - continue - } b.killLocked(e) return e } return nil } - -func (b *Buffer) drainDeadLocked() { - for b.max.Len() > 0 { - if b.max[0].alive { - break - } - heap.Pop(&b.max) - } - for b.min.Len() > 0 { - if b.min[0].alive { - break - } - heap.Pop(&b.min) - } -} diff --git a/pkg/blockprod/scheduler/buffer_bench_test.go b/pkg/blockprod/scheduler/buffer_bench_test.go new file mode 100644 index 000000000..ccdaa3af0 --- /dev/null +++ b/pkg/blockprod/scheduler/buffer_bench_test.go @@ -0,0 +1,45 @@ +package scheduler + +import ( + "encoding/binary" + "testing" + + "github.com/gagliardetto/solana-go" +) + +// BenchmarkBufferDrain measures selection from a large, already-filled TPU +// queue. Transaction decoding and queue filling are outside the timed region. +func BenchmarkBufferDrain(b *testing.B) { + const count = 120000 + for _, mixed := range []bool{false, true} { + name := "equal_rewards" + if mixed { + name = "mixed_rewards" + } + b.Run(name, func(b *testing.B) { + var buffer *Buffer + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if i%count == 0 { + b.StopTimer() + buffer = NewBuffer(count) + for j := 0; j < count; j++ { + e := &entry{tx: &solana.Transaction{}, seq: uint64(j), reward: 2500} + binary.LittleEndian.PutUint64(e.messageHash[:], uint64(j)) + if mixed { + e.reward = uint64(j*7919) % 1000 + } + if result, _ := buffer.Insert(e); result != InsertAccepted { + b.Fatal("queue fill failed") + } + } + b.StartTimer() + } + if buffer.PopMax() == nil { + b.Fatal("queue drained prematurely") + } + } + }) + } +} diff --git a/pkg/blockprod/scheduler/buffer_order_test.go b/pkg/blockprod/scheduler/buffer_order_test.go new file mode 100644 index 000000000..11a091bfb --- /dev/null +++ b/pkg/blockprod/scheduler/buffer_order_test.go @@ -0,0 +1,106 @@ +package scheduler + +import ( + "container/heap" + "encoding/binary" + "math/rand" + "sort" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestMaxHeapRemovalMatchesSortedOrder(t *testing.T) { + rng := rand.New(rand.NewSource(417)) + for _, count := range []int{1, 2, 3, 4, 7, 8, 9, 127, 128, 129, 4095, 4096, 4097} { + var h maxHeap + want := make([]*entry, count) + for i := range want { + want[i] = &entry{seq: uint64(i), reward: uint64(rng.Intn(17))} + heap.Push(&h, want[i]) + } + sort.Slice(want, func(i, j int) bool { + if want[i].reward == want[j].reward { + return want[i].seq < want[j].seq + } + return want[i].reward > want[j].reward + }) + backing := h[:cap(h)] + for i, expected := range want { + require.Same(t, expected, h.popEntry(), "count=%d pop=%d", count, i) + require.Nil(t, backing[len(h)], "removed pointer retained") + } + } +} + +// Compare interleaved insertion, eviction, cleanup and removal with a small +// unsorted reference model. Reusing hashes after removal exercises replacement +// entries as well as counterpart removal from both indexed heaps. +func TestBufferMixedOperationsMatchReference(t *testing.T) { + const capacity = 64 + rng := rand.New(rand.NewSource(418)) + b := NewBuffer(capacity) + model := make(map[[32]byte]*entry) + best := func(high bool) *entry { + var found *entry + for _, e := range model { + if found == nil || (high && (e.reward > found.reward || e.reward == found.reward && e.seq < found.seq)) || + (!high && (e.reward < found.reward || e.reward == found.reward && e.seq > found.seq)) { + found = e + } + } + return found + } + for step := 0; step < 10000; step++ { + switch action := rng.Intn(10); { + case action < 7: + e := &entry{tx: &solana.Transaction{}, seq: uint64(step), reward: uint64(rng.Intn(16))} + binary.LittleEndian.PutUint64(e.messageHash[:], uint64(rng.Intn(256))) + wantResult := InsertAccepted + var wantEvicted *entry + if _, duplicate := model[e.messageHash]; duplicate { + wantResult = InsertDuplicate + } else if len(model) == capacity { + lowest := best(false) + if e.reward <= lowest.reward { + wantResult = InsertRejectedCapacity + } else { + wantEvicted = lowest + delete(model, lowest.messageHash) + } + } + if wantResult == InsertAccepted { + model[e.messageHash] = e + } + got, evicted := b.Insert(e) + require.Equal(t, wantResult, got, "step=%d", step) + require.True(t, wantEvicted == evicted, "eviction differs at step=%d", step) + case action < 9: + want := best(true) + got := b.PopMax() + require.True(t, want == got, "selection differs at step=%d", step) + if want != nil { + delete(model, want.messageHash) + } + default: + mod := uint64(rng.Intn(11)) + want := 0 + for hash, e := range model { + if e.seq%11 == mod { + delete(model, hash) + want++ + } + } + require.Equal(t, want, b.Cleanup(func(e *entry) bool { return e.seq%11 == mod })) + } + require.Equal(t, len(model), b.Len(), "step=%d", step) + assertBufferIndexes(t, b) + } + for len(model) > 0 { + want := best(true) + require.Same(t, want, b.PopMax()) + delete(model, want.messageHash) + } + require.Nil(t, b.PopMax()) +} diff --git a/pkg/blockprod/scheduler/buffer_retention_test.go b/pkg/blockprod/scheduler/buffer_retention_test.go new file mode 100644 index 000000000..6b24aac8a --- /dev/null +++ b/pkg/blockprod/scheduler/buffer_retention_test.go @@ -0,0 +1,158 @@ +package scheduler + +import ( + "encoding/binary" + "sync" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func retainedTestEntry(id, reward uint64) *entry { + e := &entry{tx: &solana.Transaction{}, wire: []byte{1, 2, 3}, seq: id, reward: reward} + binary.LittleEndian.PutUint64(e.messageHash[:], id) + return e +} + +// Check both membership and backing-array references: shrinking a slice alone +// must not leave transaction payloads reachable outside its visible length. +func assertBufferIndexes(t *testing.T, b *Buffer) { + t.Helper() + b.mu.Lock() + defer b.mu.Unlock() + require.Equal(t, b.alive, len(b.byHash)) + require.Equal(t, b.alive, len(b.max)) + require.Equal(t, b.alive, len(b.min)) + require.LessOrEqual(t, b.alive, b.capacity) + for i, e := range b.max { + require.True(t, e.alive) + require.Equal(t, i, e.maxIndex) + require.Same(t, e, b.byHash[e.messageHash]) + require.Same(t, e, b.min[e.minIndex]) + if i > 0 { + require.False(t, b.max.Less(i, (i-1)/2)) + } + } + for i, e := range b.min { + require.Equal(t, i, e.minIndex) + require.Same(t, e, b.max[e.maxIndex]) + if i > 0 { + require.False(t, b.min.Less(i, (i-1)/2)) + } + } + for _, backing := range [][]*entry{b.max[:cap(b.max)], b.min[:cap(b.min)]} { + for _, e := range backing[b.alive:] { + require.Nil(t, e) + } + } +} + +func TestBufferConsumedEntriesReleaseBothHeapReferences(t *testing.T) { + b := NewBuffer(256) + pinned := retainedTestEntry(0, 1) + b.Insert(pinned) + for i := uint64(1); i <= 100000; i++ { + e := retainedTestEntry(i, 2) + result, _ := b.Insert(e) + require.Equal(t, InsertAccepted, result) + require.Same(t, e, b.PopMax()) + // The consumer still owns usable payloads after index removal. + require.NotNil(t, e.tx) + require.Equal(t, []byte{1, 2, 3}, e.wire) + if i%1000 == 0 { + assertBufferIndexes(t, b) + } + } + b.Cleanup(func(*entry) bool { return false }) + assertBufferIndexes(t, b) + t.Logf("capacity=%d active=%d max_refs=%d min_refs=%d", b.capacity, b.Len(), len(b.max), len(b.min)) + require.Same(t, pinned, b.PopMax()) + assertBufferIndexes(t, b) +} + +func TestBufferEvictionAndCleanupReleaseBothHeapReferences(t *testing.T) { + b := NewBuffer(2) + pinned := retainedTestEntry(0, 1000000) + b.Insert(pinned) + previous := retainedTestEntry(1, 1) + b.Insert(previous) + for i := uint64(2); i < 10000; i++ { + next := retainedTestEntry(i, i) + result, evicted := b.Insert(next) + require.Equal(t, InsertAccepted, result) + require.Same(t, previous, evicted) + require.False(t, evicted.alive) + require.Equal(t, -1, evicted.maxIndex) + require.Equal(t, -1, evicted.minIndex) + previous = next + if i%100 == 0 { + assertBufferIndexes(t, b) + } + } + require.Equal(t, 1, b.Cleanup(func(e *entry) bool { return e != pinned })) + assertBufferIndexes(t, b) + require.Same(t, pinned, b.PopMax()) + assertBufferIndexes(t, b) +} + +func TestBufferRepeatedRebufferPreservesNewHigherPriorityArrivals(t *testing.T) { + s := New(nil) + s.bankGen = 1 + skipped := retainedTestEntry(1, 10) + skipped.skipGen = s.bankGen + s.buffer.Insert(skipped) + for i := uint64(2); i < 10002; i++ { + low := retainedTestEntry(i, 1) + s.buffer.Insert(low) + picked, retry := s.popSchedulable(s.bankGen) + require.Same(t, low, picked) + require.Equal(t, []*entry{skipped}, retry) + for _, e := range retry { + s.rebuffer(e) + } + if i%100 == 0 { + assertBufferIndexes(t, s.buffer) + } + } + // Preserve the existing scan/retry policy when a higher-fee packet arrives. + high := retainedTestEntry(20000, 20) + s.buffer.Insert(high) + picked, retry := s.popSchedulable(s.bankGen) + require.Same(t, high, picked) + require.Empty(t, retry) + // A later bank may retry the previously skipped transaction. + s.bankGen++ + picked, retry = s.popSchedulable(s.bankGen) + require.Same(t, skipped, picked) + require.Empty(t, retry) + assertBufferIndexes(t, s.buffer) +} + +func TestBufferConcurrentInsertRemovalAndCleanup(t *testing.T) { + b := NewBuffer(64) + var workers sync.WaitGroup + for worker := uint64(0); worker < 4; worker++ { + workers.Go(func() { + for i := uint64(0); i < 2000; i++ { + id := worker*2000 + i + b.Insert(retainedTestEntry(id, id%17)) + } + }) + } + workers.Go(func() { + for i := 0; i < 8000; i++ { + b.PopMax() + } + }) + workers.Go(func() { + for i := 0; i < 100; i++ { + b.Cleanup(func(e *entry) bool { return e.seq%3 == 0 }) + } + }) + workers.Wait() + assertBufferIndexes(t, b) + for b.PopMax() != nil { + } + assertBufferIndexes(t, b) +} diff --git a/pkg/blockprod/scheduler/buffer_test.go b/pkg/blockprod/scheduler/buffer_test.go index 9b85554ce..6ec46415d 100644 --- a/pkg/blockprod/scheduler/buffer_test.go +++ b/pkg/blockprod/scheduler/buffer_test.go @@ -79,3 +79,28 @@ func TestBufferCleanup(t *testing.T) { require.Equal(t, 1, b.Len()) require.Equal(t, byte(2), b.PopMax().messageHash[0]) } + +func TestBufferPrecheckDoesNotReserveAndInsertRechecks(t *testing.T) { + b := NewBuffer(1) + first := testEntry(10, 1, 1) + require.Equal(t, InsertAccepted, b.precheck(first)) + require.Zero(t, b.Len()) + b.Insert(first) + require.Equal(t, InsertDuplicate, b.precheck(testEntry(20, 2, 1))) + require.Equal(t, InsertRejectedCapacity, b.precheck(testEntry(10, 2, 2))) + candidate := testEntry(20, 3, 3) + require.Equal(t, InsertAccepted, b.precheck(candidate)) + // A precheck must not evict the existing entry, and a later higher-priority + // arrival can reject the candidate even though its precheck succeeded. + require.Equal(t, first, b.PopMax()) + b.Insert(testEntry(30, 4, 4)) + result, evicted := b.Insert(candidate) + require.Equal(t, InsertRejectedCapacity, result) + require.Nil(t, evicted) + b.PopMax() + require.Equal(t, InsertAccepted, b.precheck(candidate)) + b.Insert(testEntry(20, 5, 3)) + result, evicted = b.Insert(candidate) + require.Equal(t, InsertDuplicate, result) + require.Nil(t, evicted) +} diff --git a/pkg/blockprod/scheduler/config_test.go b/pkg/blockprod/scheduler/config_test.go new file mode 100644 index 000000000..9f43ed511 --- /dev/null +++ b/pkg/blockprod/scheduler/config_test.go @@ -0,0 +1,28 @@ +package scheduler + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/stretchr/testify/require" +) + +func TestConfiguredQueueCapacityAndPreparation(t *testing.T) { + defaults := NewWithConfig(nil, Config{}) + require.Equal(t, MaxBufferedTxns, defaults.buffer.Capacity()) + feats := features.NewFeaturesDefault() + custom := NewWithConfig(nil, Config{MaxBufferedTransactions: 2, FeatureSource: func() *features.Features { return feats }}) + require.Equal(t, 2, custom.buffer.Capacity()) + require.NotNil(t, custom.preparer.Load()) + for i := byte(1); i <= 2; i++ { + result, _ := custom.buffer.Insert(testEntry(10, uint64(i), i)) + require.Equal(t, InsertAccepted, result) + } + result, evicted := custom.buffer.Insert(testEntry(9, 3, 3)) + require.Equal(t, InsertRejectedCapacity, result) + require.Nil(t, evicted) + result, evicted = custom.buffer.Insert(testEntry(11, 4, 4)) + require.Equal(t, InsertAccepted, result) + require.Equal(t, uint64(2), evicted.seq) + require.Equal(t, 2, custom.Buffered()) +} diff --git a/pkg/blockprod/scheduler/scheduler.go b/pkg/blockprod/scheduler/scheduler.go index de6196f30..9b0864d27 100644 --- a/pkg/blockprod/scheduler/scheduler.go +++ b/pkg/blockprod/scheduler/scheduler.go @@ -9,6 +9,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/blockprod" "github.com/Overclock-Validator/mithril/pkg/costmodel" "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/replay" "github.com/Overclock-Validator/mithril/pkg/tpu/packet" "github.com/gagliardetto/solana-go" ) @@ -43,9 +44,12 @@ type Stats struct { // Scheduler buffers verified TPU transactions in a reward-ordered heap and // drains into the active WorkingBank when one is published. type Scheduler struct { - banks BankSource - feats *features.Features - buffer *Buffer + banks BankSource + feats *features.Features + buffer *Buffer + preparer atomic.Pointer[replay.TransactionPreparer] + featureSource func() *features.Features + lastPreparationRefresh time.Time seq atomic.Uint64 wake chan struct{} @@ -80,6 +84,37 @@ func New(banks BankSource) *Scheduler { } } +// NewWithFeatureSource prepares queued messages against a replay-tip snapshot. +// The active bank independently verifies feature compatibility before reuse. +func NewWithFeatureSource(banks BankSource, source func() *features.Features) *Scheduler { + return NewWithConfig(banks, Config{FeatureSource: source}) +} + +// Config controls the bounded TPU queue and static preparation source. +type Config struct { + // MaxBufferedTransactions defaults to MaxBufferedTxns when zero. + MaxBufferedTransactions int + FeatureSource func() *features.Features +} + +func NewWithConfig(banks BankSource, cfg Config) *Scheduler { + s := New(banks) + if cfg.MaxBufferedTransactions > 0 { + s.buffer = NewBuffer(cfg.MaxBufferedTransactions) + } + s.featureSource = cfg.FeatureSource + s.refreshPreparation() + return s +} + +func (s *Scheduler) refreshPreparation() { + if s.featureSource == nil || time.Since(s.lastPreparationRefresh) < 100*time.Millisecond { + return + } + s.preparer.Store(replay.NewTransactionPreparer(s.featureSource())) + s.lastPreparationRefresh = time.Now() +} + // Start launches the bank-gated drain loop. func (s *Scheduler) Start(ctx context.Context) { s.startOnce.Do(func() { @@ -145,7 +180,12 @@ func (s *Scheduler) Receive(pkt packet.Packet) { reward: reward, seq: s.seq.Add(1), } - result, evicted := s.buffer.Insert(e) + result := s.buffer.precheck(e) + var evicted *entry + if result == InsertAccepted { + e.prepared = s.preparer.Load().Prepare(tx) + result, evicted = s.buffer.Insert(e) + } s.mu.Lock() switch result { case InsertAccepted: @@ -219,6 +259,7 @@ func (s *Scheduler) drainLoop(ctx context.Context) { if ctx.Err() != nil { return } + s.refreshPreparation() bank := s.banks.WorkingBank() s.noteBank(bank) if bank == nil { @@ -258,15 +299,10 @@ func (s *Scheduler) drainLoop(ctx context.Context) { continue } - // Prefer the owned wire so forge reparses from stable bytes even if the - // retained tx view was somehow mutated after buffering. - var result blockprod.ForgeResult - var reason costmodel.ExceedReason - if len(e.wire) > 0 { - result, reason = bank.Forge(e.wire) - } else { - result, reason = bank.ForgeTransaction(e.tx, e.wireSize) - } + // Receive owns the wire and its decoded transaction for the entire queue + // lifetime. Execution modifies transaction-local account clones, not + // this immutable message, so reuse the decoded transaction across banks. + result, reason := bank.ForgePreparedTransaction(e.tx, e.wireSize, e.prepared) switch result { case blockprod.ForgeDroppedNoLeader: s.rebuffer(e) diff --git a/pkg/blockprod/scheduler/scheduler_test.go b/pkg/blockprod/scheduler/scheduler_test.go index e4b74e092..f83711312 100644 --- a/pkg/blockprod/scheduler/scheduler_test.go +++ b/pkg/blockprod/scheduler/scheduler_test.go @@ -7,6 +7,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/blockprod" "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" "github.com/Overclock-Validator/mithril/pkg/fees" "github.com/Overclock-Validator/mithril/pkg/tpu/packet" "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" @@ -244,7 +245,9 @@ func TestMaxBufferedTxnsConstant(t *testing.T) { } func TestSchedulerCopiesPooledPacketBytes(t *testing.T) { - sched := New(blockprod.NewController()) + env := blockprod.NewTestEnv(blockprod.TestEnvConfig{}) + defer env.Close() + sched := NewWithFeatureSource(blockprod.NewController(), func() *features.Features { return env.SlotCtx.Features.Clone() }) pool := packet.NewPool(1) buf, idx, ok := pool.Acquire() require.True(t, ok) @@ -269,11 +272,26 @@ func TestSchedulerCopiesPooledPacketBytes(t *testing.T) { e := sched.buffer.PopMax() require.NotNil(t, e) require.Equal(t, wire, e.wire) + require.NotNil(t, e.prepared) // Re-parse and verify the buffered transaction still has a valid signature. tx, err := solana.TransactionFromBytes(e.wire) require.NoError(t, err) require.Equal(t, e.tx.Signatures[0], tx.Signatures[0]) + + // Exercise the actual drain path using the retained decoded transaction, + // after its original pooled packet has already been overwritten. + res, _ := sched.buffer.Insert(e) + require.Equal(t, InsertAccepted, res) + sched.banks = env.Controller + sched.Start(context.Background()) + defer sched.Stop() + require.Eventually(t, func() bool { return sched.Stats().Accepted == 1 }, time.Second, time.Millisecond) + forged := env.Bank.ForgedTransactions() + require.Len(t, forged, 1) + forgedWire, err := forged[0].MarshalBinary() + require.NoError(t, err) + require.Equal(t, wire, forgedWire) } func TestClassifyBufferedExpired(t *testing.T) { diff --git a/pkg/blockstream/block_source.go b/pkg/blockstream/block_source.go index 49212382c..6c9b6f4e1 100644 --- a/pkg/blockstream/block_source.go +++ b/pkg/blockstream/block_source.go @@ -47,18 +47,21 @@ type BlockSourceOpts struct { // Enables Alpenglow/Votor block-id hints for the Turbine assembler. Classic // Solana clusters leave this off even when blocks are sourced from Turbine. TurbineAlpenglowBlockIDHints bool - TurbineIdentity ed25519.PrivateKey - LeaderForSlot func(slot uint64) (solana.PublicKey, bool) - TurbineStakesForSlot func(slot uint64) map[solana.PublicKey]uint64 - TurbineEpochForSlot func(slot uint64) uint64 - TurbineRootSlot func() uint64 - TurbineUseChaCha8 bool - TurbineDedupAddrs bool - LocalLeaderForSlot func(slot uint64) bool - GossipClient *gossip.Client - AlpenglowDecisionSource func(anchorSlot uint64) (alpenglow.ChainDecision, bool) - AlpenglowCandidateBlockSink func(alpenglow.ReplayBlockObservation) - AlpenglowInvalidBlockSink func(alpenglow.BlockID, string) error + // TurbineStreamingExecution subscribes replay's streaming executor to the + // assembler's decoded-batch feed (StreamEvents). Off by default. + TurbineStreamingExecution bool + TurbineIdentity ed25519.PrivateKey + LeaderForSlot func(slot uint64) (solana.PublicKey, bool) + TurbineStakesForSlot func(slot uint64) map[solana.PublicKey]uint64 + TurbineEpochForSlot func(slot uint64) uint64 + TurbineRootSlot func() uint64 + TurbineUseChaCha8 bool + TurbineDedupAddrs bool + LocalLeaderForSlot func(slot uint64) bool + GossipClient *gossip.Client + AlpenglowDecisionSource func(anchorSlot uint64) (alpenglow.ChainDecision, bool) + AlpenglowCandidateBlockSink func(alpenglow.ReplayBlockObservation) + AlpenglowInvalidBlockSink func(alpenglow.BlockID, string) error // AlpenglowCandidateValidator prevents objectively invalid assembled blocks // from polluting the early ancestry tracker. Replay independently validates // again at the consensus boundary before observing or executing the block. @@ -447,6 +450,11 @@ type BlockSource struct { knownAlpenglowBlockIDs map[uint64]solana.Hash knownAlpenglowBlockIDOrder []uint64 activeTurbineReceiver *turbine.UDPReceiver + // streamEvents carries the turbine streaming feed to replay; nil unless + // TurbineStreamingExecution was requested. Sized for several blocks of + // batches; a full channel drops wake-ups, which the consumer tolerates by + // polling PendingStreamBatches. + streamEvents chan turbine.StreamEvent // Repair-first catchup: gap slots [repairCatchupFrom, repairCatchupUntil] // fill via turbine repair; RPC never fetches at/above the gate while // pending or active. The pending hold persists from construction until @@ -766,6 +774,7 @@ func NewBlockSource(opts *BlockSourceOpts) *BlockSource { turbineShredVersion: opts.TurbineShredVersion, turbineAlpenglowAddr: opts.TurbineAlpenglowAddr, turbineAlpenglowBlockIDHints: opts.TurbineAlpenglowBlockIDHints, + streamEvents: newStreamEventChannel(opts), turbineIdentity: clonePrivateKey(opts.TurbineIdentity), leaderForSlot: opts.LeaderForSlot, turbineStakesForSlot: opts.TurbineStakesForSlot, @@ -3775,20 +3784,48 @@ func (bs *BlockSource) NextBlock() *b.Block { // may be nil to disable decision wakeups, but must never be closed. The third // result distinguishes a decision wakeup from a closed stream or cancellation. func (bs *BlockSource) NextBlockOrAlpenglowEvent(ctx context.Context, decisionChanges <-chan struct{}) (block *b.Block, parentSwitch *AlpenglowParentSwitch, decisionChanged bool) { + in := bs.NextReplayInput(ctx, decisionChanges, nil, nil) + return in.Block, in.ParentSwitch, in.DecisionChanged +} + +// ReplayInput is one wake-up of replay's wait for the next thing to do. At +// most one field is set; the zero value means the wait context ended or the +// source closed. +type ReplayInput struct { + Block *b.Block + ParentSwitch *AlpenglowParentSwitch + DecisionChanged bool + // StreamEvent is a turbine streaming-feed wake-up (see StreamEvents). + StreamEvent *turbine.StreamEvent + // StreamTick is the streaming executor's poll timer. + StreamTick bool +} + +// NextReplayInput is NextBlockOrAlpenglowEvent extended with the streaming +// feed and the executor's poll timer, both of which may be nil (never fire). +// A queued parent switch keeps its priority; among the remaining inputs the +// choice is the runtime's, which is fine because every stream wake-up is +// idempotent and the complete block is processed the same way whether or not +// the feed was drained first. +func (bs *BlockSource) NextReplayInput(ctx context.Context, decisionChanges <-chan struct{}, streamEvents <-chan turbine.StreamEvent, streamTick <-chan time.Time) ReplayInput { select { case event := <-bs.alpenglowParentSwitchCh: - return nil, &event, false + return ReplayInput{ParentSwitch: &event} default: } select { case event := <-bs.alpenglowParentSwitchCh: - return nil, &event, false + return ReplayInput{ParentSwitch: &event} case block := <-bs.streamChan: - return block, nil, false + return ReplayInput{Block: block} case <-decisionChanges: - return nil, nil, true + return ReplayInput{DecisionChanged: true} + case event := <-streamEvents: + return ReplayInput{StreamEvent: &event} + case <-streamTick: + return ReplayInput{StreamTick: true} case <-ctx.Done(): - return nil, nil, false + return ReplayInput{} } } diff --git a/pkg/blockstream/turbine_stream.go b/pkg/blockstream/turbine_stream.go index 1d1228c45..5e53b32a9 100644 --- a/pkg/blockstream/turbine_stream.go +++ b/pkg/blockstream/turbine_stream.go @@ -221,6 +221,9 @@ func (bs *BlockSource) attachAlpenglowBlockIDHintsToReceiver(receiver *turbine.U // would then make its slot permanently unfetchable. bs.alpenglowMu.Lock() bs.activeTurbineReceiver = receiver + if bs.streamEvents != nil { + receiver.SubscribeStream(bs.streamEvents) + } if !bs.turbineAlpenglowBlockIDHints { bs.alpenglowMu.Unlock() return @@ -568,3 +571,63 @@ func (bs *BlockSource) runTurbineStream() { } } } + +// streamEventBuffer bounds the streaming feed: a heavy block is a few hundred +// DATA_COMPLETE ranges, and the consumer drains between groups. +const streamEventBuffer = 4096 + +func newStreamEventChannel(opts *BlockSourceOpts) chan turbine.StreamEvent { + if opts == nil || opts.SourceType != BlockSourceTurbine || !opts.TurbineStreamingExecution { + return nil + } + return make(chan turbine.StreamEvent, streamEventBuffer) +} + +// StreamEvents is the turbine streaming feed for replay's streaming executor; +// nil when streaming execution is not enabled for this source. +func (bs *BlockSource) StreamEvents() <-chan turbine.StreamEvent { + if bs.streamEvents == nil { + return nil + } + return bs.streamEvents +} + +// StreamStatusOf reports the assembler's view of a streaming generation +// through the active receiver; a generation is gone when no receiver is +// active. +func (bs *BlockSource) StreamStatusOf(g turbine.StreamGeneration) turbine.StreamStatus { + bs.alpenglowMu.Lock() + receiver := bs.activeTurbineReceiver + bs.alpenglowMu.Unlock() + if receiver == nil { + return turbine.StreamGone + } + return receiver.StreamStatusOf(g) +} + +// PendingStreamBatches returns the generation's decoded batches at or after +// fromStart, in shred-index order, from the active receiver. +func (bs *BlockSource) PendingStreamBatches(g turbine.StreamGeneration, fromStart uint32) []*turbine.StreamBatch { + bs.alpenglowMu.Lock() + receiver := bs.activeTurbineReceiver + bs.alpenglowMu.Unlock() + if receiver == nil { + return nil + } + return receiver.PendingStreamBatches(g, fromStart) +} + +// PrioritizeStreamRepair keeps a slot that replay is executing while its +// shreds arrive pinned for repair, since the emitter pins the head only when +// it observes a gap. +func (bs *BlockSource) PrioritizeStreamRepair(g turbine.StreamGeneration) { + if bs.sourceType != BlockSourceTurbine || !bs.turbineAlpenglowBlockIDHints { + return + } + bs.alpenglowMu.Lock() + receiver := bs.activeTurbineReceiver + bs.alpenglowMu.Unlock() + if receiver != nil { + receiver.PrioritizeStreamRepair(g) + } +} diff --git a/pkg/config/config.go b/pkg/config/config.go index 1e7416c90..501dfa158 100644 --- a/pkg/config/config.go +++ b/pkg/config/config.go @@ -25,6 +25,7 @@ func ApplyDefaults(v *viper.Viper) { v.SetDefault("validator.tpu_quic_bind_addr", "") v.SetDefault("validator.advertised_ip", "") v.SetDefault("validator.tpu_sigverify_workers", 0) + v.SetDefault("validator.wait_to_vote_slot", uint64(0)) } // LedgerConfig holds ledger-related configuration (matches Firedancer [ledger] section) @@ -235,6 +236,9 @@ type ValidatorConfig struct { TPUQUICBindAddr string `toml:"tpu_quic_bind_addr" mapstructure:"tpu_quic_bind_addr"` AdvertisedIP string `toml:"advertised_ip" mapstructure:"advertised_ip"` TPUSigverifyWorkers int `toml:"tpu_sigverify_workers" mapstructure:"tpu_sigverify_workers"` + WaitToVoteSlot uint64 `toml:"wait_to_vote_slot" mapstructure:"wait_to_vote_slot"` // Minimum slot for new votes; automatic startup cutoff still applies + BlockCompletionReserveMs int `toml:"block_completion_reserve_ms" mapstructure:"block_completion_reserve_ms"` + TPUMaxBufferedTransactions int `toml:"tpu_max_buffered_transactions" mapstructure:"tpu_max_buffered_transactions"` } // Config holds all configuration options for Mithril (Firedancer-style hierarchy) diff --git a/pkg/consensus/engine.go b/pkg/consensus/engine.go index f57cb2ab5..5cf2cb203 100644 --- a/pkg/consensus/engine.go +++ b/pkg/consensus/engine.go @@ -389,7 +389,7 @@ func (e *AlpenglowObserverEngine) EnableVoting(cfg VotingConfig) error { // events, matching Agave's initial_parent_ready selection. if slot, parent, ok := voter.history.HighestParentReadyMatching(func(parent alpenglow.BlockID) bool { return !e.ensureChain().IsObjectivelyInvalidBlock(parent) - }); ok && slot > root.Slot { + }); ok && slot > root.Slot && (voter.reservation == nil || voter.reservation.recoverThrough == 0) { if !e.ensurePool().RestoreParentReady(slot, parent) { mlog.Log.FileOnlyf("ALPENGLOW voting: ignored persisted ParentReady slot=%d parent=%s because newer root/live tracker state is authoritative", slot, parent) } @@ -975,11 +975,18 @@ func (e *AlpenglowObserverEngine) injectLocalVote(message alpenglow.VoteMessage, } } -// alpenglowVoteActionFloor is the highest slot on which this validator must -// not initiate a new vote. The pool root is its strict admission boundary; -// direct finality is included because the pool deliberately retains a short -// reward-accounting tail behind finality where network votes remain useful. +// alpenglowVoteActionFloor is the retained pool's strict admission boundary. +// Network finality is not a voting root: a replayed block may still contribute +// a notarization to a later fast certificate and its slot+8 reward certificate. +// The voter also checks its own persisted history root before signing. func (e *AlpenglowObserverEngine) alpenglowVoteActionFloor() uint64 { + return e.ensurePool().Snapshot().RootSlot +} + +// alpenglowVerifiedFinalityFloor releases crash-recovery reservations. Keep this +// independent of live vote admission: a retained reward window must not weaken +// the requirement to pass every slot that may have been signed before a crash. +func (e *AlpenglowObserverEngine) alpenglowVerifiedFinalityFloor() uint64 { floor := e.ensurePool().Snapshot().RootSlot if finalized := e.ensureChain().Snapshot().LatestDirectFinalizedBlock.Slot; finalized > floor { floor = finalized @@ -1542,6 +1549,25 @@ func (e *AlpenglowObserverEngine) PruneAlpenglowBefore(slot uint64) { if slot == 0 { return } + // Replay enqueues its completed-block event before publishing a durable + // promotion. Retire the pool, execution proof and history on that same + // ordered voter stream, so a fast checkpoint cannot overtake the vote. + e.voterMu.RLock() + voter := e.voter + e.voterMu.RUnlock() + if voter != nil { + if err := voter.enqueue(voterEvent{kind: voterEventDurableRoot, slot: slot}); err != nil { + e.latchSafetyError(err) + } + return + } + e.applyAlpenglowDurableRoot(slot) +} + +// applyAlpenglowDurableRoot is called by the voter after earlier replay events, +// or synchronously by an observer without a voting loop. Startup root restore +// remains a separate, immediate barrier in SetAlpenglowRoot. +func (e *AlpenglowObserverEngine) applyAlpenglowDurableRoot(slot uint64) alpenglow.BlockID { e.poolOutputMu.Lock() defer e.poolOutputMu.Unlock() @@ -1557,13 +1583,11 @@ func (e *AlpenglowObserverEngine) PruneAlpenglowBefore(slot uint64) { if e.certPool != nil { e.certPool.ObserveFloor(slot) } - if err := e.enqueueVoter(voterEvent{kind: voterEventRoot, root: root}); err != nil { - e.latchSafetyError(err) - } // Replay calls this only after the fold through slot is durably committed. // Keep the transport peer window tied to that local root, not to speculative // certificate finality or the pool's reward-retention floor. e.advanceVotorPeerRoot(slot) + return root } func (e *AlpenglowObserverEngine) pruneInvalidBlockIDsBefore(slot uint64) { diff --git a/pkg/consensus/vote_history_writer.go b/pkg/consensus/vote_history_writer.go new file mode 100644 index 000000000..9d175d6d4 --- /dev/null +++ b/pkg/consensus/vote_history_writer.go @@ -0,0 +1,125 @@ +package consensus + +import ( + "errors" + "fmt" + "sync" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" +) + +// One serial writer, one in-flight snapshot and at most one newer pending +// snapshot. A newer complete history supersedes an unwritten snapshot; every +// retained, unrooted voting decision is still present in that newer history. +// The independently durable reservation, not this queue, authorizes signing. +// In-flight/pending snapshots may be lost on process death; even a completed +// unsynced replacement may be lost on host/power failure. Neither submitted nor +// written is a durable vote acknowledgement. Recovery must use the startup +// reservation unless the separate clean-history seal validates. +type voteHistoryWriter struct { + mu sync.Mutex + pending *alpenglow.VoteHistorySnapshot + closing bool + failure error + submitted uint64 + written uint64 + coalesced uint64 + wake chan struct{} + done chan struct{} + persist func(*alpenglow.VoteHistorySnapshot) error + onError func(error) +} + +func newVoteHistoryWriter(persist func(*alpenglow.VoteHistorySnapshot) error, onError func(error)) *voteHistoryWriter { + w := &voteHistoryWriter{wake: make(chan struct{}, 1), done: make(chan struct{}), persist: persist, onError: onError} + go w.run() + return w +} + +// submit does no I/O and never waits for the writer. The mutex only protects +// pointer/counter changes; neither persistence nor error callbacks hold it. +// A nil return means queued only. It must never replace the reservation check. +func (w *voteHistoryWriter) submit(snapshot *alpenglow.VoteHistorySnapshot) error { + if snapshot == nil { + return errors.New("nil vote-history snapshot") + } + w.mu.Lock() + if w.failure != nil { + err := w.failure + w.mu.Unlock() + return err + } + if w.closing { + w.mu.Unlock() + return errors.New("vote-history writer is closed") + } + if w.pending != nil { + w.coalesced++ + } + w.pending = snapshot + w.submitted++ + w.mu.Unlock() + w.notify() + return nil +} + +func (w *voteHistoryWriter) notify() { + select { + case w.wake <- struct{}{}: + default: + } +} + +func (w *voteHistoryWriter) run() { + defer close(w.done) + for range w.wake { + for { + w.mu.Lock() + snapshot := w.pending + w.pending = nil + closing := w.closing + w.mu.Unlock() + if snapshot == nil { + if closing { + return + } + break + } + if err := w.persist(snapshot); err != nil { + err = fmt.Errorf("background vote-history write: %w", err) + w.mu.Lock() + w.failure = err + w.pending = nil + w.closing = true + w.mu.Unlock() + if w.onError != nil { + w.onError(err) + } + return + } + w.mu.Lock() + w.written++ + w.mu.Unlock() + } + } +} + +// close rejects new submissions and drains every retained snapshot. The voter +// must join this worker before writing and syncing its final clean history, +// otherwise an older in-flight rename could overwrite the sealed history. +func (w *voteHistoryWriter) close() error { + w.mu.Lock() + w.closing = true + w.mu.Unlock() + w.notify() + <-w.done + w.mu.Lock() + defer w.mu.Unlock() + return w.failure +} + +func (w *voteHistoryWriter) counters() (submitted, written, coalesced uint64) { + w.mu.Lock() + defer w.mu.Unlock() + return w.submitted, w.written, w.coalesced +} diff --git a/pkg/consensus/vote_history_writer_test.go b/pkg/consensus/vote_history_writer_test.go new file mode 100644 index 000000000..74c7bbfcd --- /dev/null +++ b/pkg/consensus/vote_history_writer_test.go @@ -0,0 +1,201 @@ +package consensus + +import ( + "errors" + "os" + "os/exec" + "path/filepath" + "sync" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestAsyncHistoryBlockedWriteDoesNotDelayVotesAndCleanCloseDrains(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + require.NoError(t, v.historyWriter.close()) + entered, release := make(chan struct{}), make(chan struct{}) + var once sync.Once + unblock := func() { once.Do(func() { close(release) }) } + t.Cleanup(unblock) + first := true // Owned only by the serial writer. + v.historyWriter = newVoteHistoryWriter(func(s *alpenglow.VoteHistorySnapshot) error { + if first { + first = false + close(entered) + <-release + } + return alpenglow.SaveReservedVoteHistorySnapshot(cfg.HistoryDir, s) + }, nil) + voted, err := v.cast(alpenglow.NewSkipVote(44), false) + require.NoError(t, err) + require.True(t, voted) + <-entered + castDone := make(chan error, 1) + go func() { + for _, slot := range []uint64{45, 46} { + ok, err := v.cast(alpenglow.NewSkipVote(slot), false) + if err != nil || !ok { + castDone <- errors.New("vote failed while history writer was blocked") + return + } + } + castDone <- nil + }() + select { + case err := <-castDone: + require.NoError(t, err) + case <-time.After(time.Second): + t.Fatal("disk writer blocked voting") + } + submitted, written, coalesced := v.historyWriter.counters() + require.Equal(t, uint64(3), submitted) + require.Zero(t, written) + require.Equal(t, uint64(1), coalesced) + onDisk, err := alpenglow.LoadVoteHistory(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.False(t, onDisk.HasSkipped(44), "I/O should still be blocked") + // Even when disk history lags, complete in-memory decisions forbid conflict. + voted, err = v.cast(alpenglow.NewNotarizationVote(44, solana.Hash{7}), false) + require.NoError(t, err) + require.False(t, voted) + closed := make(chan error, 1) + go func() { closed <- v.close() }() + require.Eventually(t, func() bool { + v.historyWriter.mu.Lock() + defer v.historyWriter.mu.Unlock() + return v.historyWriter.closing + }, time.Second, time.Millisecond) + select { + case err := <-closed: + t.Fatalf("close returned before draining its writer: %v", err) + default: + } + r, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Empty(t, r.CleanHistoryDigest) + unblock() + require.NoError(t, <-closed) + onDisk, err = alpenglow.LoadVoteHistory(cfg.HistoryDir, v.node) + require.NoError(t, err) + for _, slot := range []uint64{44, 45, 46} { + require.True(t, onDisk.HasSkipped(slot)) + } + r, err = alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + digest, err := alpenglow.VoteHistoryDigest(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Equal(t, digest, r.CleanHistoryDigest) + _, written, _ = v.historyWriter.counters() + require.Equal(t, uint64(2), written, "old in-flight snapshot must finish before newest complete snapshot") +} + +func TestAsyncHistoryFailureIsStickyAndReportedWithoutMoreSubmissions(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + h := alpenglow.NewVoteHistory(voterTestValidatorSet(t, cfg.Identity, cfg.AuthorizedVoter, cfg.VoteAccount).Validators[0].NodePubkey, 39) + h.ReservationRequired = true + snapshot, err := alpenglow.PrepareReservedVoteHistory(h, cfg.Identity) + require.NoError(t, err) + diskErr := errors.New("injected disk failure") + reported := make(chan error, 1) + w := newVoteHistoryWriter(func(*alpenglow.VoteHistorySnapshot) error { return diskErr }, func(err error) { reported <- err }) + require.NoError(t, w.submit(snapshot)) + select { + case err := <-reported: + require.ErrorIs(t, err, diskErr) + case <-time.After(time.Second): + t.Fatal("background failure was not reported") + } + require.ErrorIs(t, w.submit(snapshot), diskErr) + require.ErrorIs(t, w.close(), diskErr) + require.ErrorIs(t, w.close(), diskErr) +} + +func TestAsyncHistoryFailureStopsVoterAndPreventsCleanMarker(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + e, err := NewEngine(Config{AlpenglowIdentity: cfg.Identity, AlpenglowShredVersion: 0x1234}) + require.NoError(t, err) + t.Cleanup(func() { _ = e.Close() }) + set := voterTestValidatorSet(t, cfg.Identity, cfg.AuthorizedVoter, cfg.VoteAccount) + e.SetAlpenglowEpochLookup(cfg.EpochForSlot) + require.NoError(t, e.SetAlpenglowValidatorSet(set)) + root := alpenglow.BlockID{Slot: 39, Hash: solana.Hash{39}} + e.SetAlpenglowRoot(root) + v, err := newAlpenglowVoterUnstarted(e, cfg, root, []alpenglow.ValidatorSet{set}) + require.NoError(t, err) + t.Cleanup(func() { _ = v.close() }) + reserveThrough(t, v.reservation, 44) + require.NoError(t, v.historyWriter.close()) + diskErr := errors.New("injected background disk failure") + v.historyWriter = newVoteHistoryWriter(func(*alpenglow.VoteHistorySnapshot) error { return diskErr }, v.failHistoryWrite) + require.NoError(t, v.history.AddVote(alpenglow.NewSkipVote(44))) + require.NoError(t, v.saveHistory()) + select { + case <-v.done: + case <-time.After(time.Second): + t.Fatal("disk failure did not stop the voter") + } + require.ErrorIs(t, e.safetyError(), diskErr) + _, _, err = v.sign(alpenglow.NewSkipVote(45), false) + require.ErrorIs(t, err, diskErr) + require.Error(t, v.enqueue(voterEvent{kind: voterEventBlockTimeout, slot: 45})) + require.ErrorIs(t, v.close(), diskErr) + r, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Empty(t, r.CleanHistoryDigest) +} + +// Kill a subprocess while its first history write is blocked and a newer +// snapshot is pending. Successful earlier reservation syncs survive this +// process crash; this is deliberately not a host power-loss test. +func TestAsyncHistoryProcessCrashLosesPendingSnapshots(t *testing.T) { + const childEnv = "MITHRIL_ASYNC_HISTORY_TEST_DIR" + if dir := os.Getenv(childEnv); dir != "" { + cfg := reservedTestConfig(dir) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + require.NoError(t, v.historyWriter.close()) + entered := make(chan struct{}) + v.historyWriter = newVoteHistoryWriter(func(*alpenglow.VoteHistorySnapshot) error { + close(entered) + select {} + }, nil) + for _, slot := range []uint64{44, 45} { + voted, err := v.cast(alpenglow.NewSkipVote(slot), false) + require.NoError(t, err) + require.True(t, voted) + if slot == 44 { + <-entered + } + } + require.NoError(t, os.WriteFile(filepath.Join(dir, "ready"), []byte("ready"), 0600)) + select {} + } + dir := t.TempDir() + cmd := exec.Command(os.Args[0], "-test.run=^TestAsyncHistoryProcessCrashLosesPendingSnapshots$", "-test.count=1") + cmd.Env = append(os.Environ(), childEnv+"="+dir) + require.NoError(t, cmd.Start()) + t.Cleanup(func() { _ = cmd.Process.Kill() }) + require.Eventually(t, func() bool { _, err := os.Stat(filepath.Join(dir, "ready")); return err == nil }, 10*time.Second, time.Millisecond) + require.NoError(t, cmd.Process.Kill()) + require.Error(t, cmd.Wait()) + cfg := reservedTestConfig(dir) + cfg.InitializeVoteReservation = false + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.False(t, v.history.HasSkipped(44)) + require.False(t, v.history.HasSkipped(45)) + h := v.reservation.recoverThrough + require.GreaterOrEqual(t, h, uint64(45)) + _, _, err = v.sign(alpenglow.NewSkipVote(44), false) + require.ErrorIs(t, err, errVoterNotReady) + _, _, err = v.sign(alpenglow.NewSkipVote(h+1), false) + require.ErrorIs(t, err, errVoterNotReady, "verified finality must reach the lost history's bound") +} diff --git a/pkg/consensus/vote_reservation.go b/pkg/consensus/vote_reservation.go new file mode 100644 index 000000000..564cfc29d --- /dev/null +++ b/pkg/consensus/vote_reservation.go @@ -0,0 +1,296 @@ +package consensus + +import ( + "bytes" + "crypto/ed25519" + "errors" + "fmt" + "math" + "os" + "sync" + "sync/atomic" + "time" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/gagliardetto/solana-go" +) + +const signingReserveSlots = uint64(32) +const signingRenewRemaining = uint64(16) + +// signingReservation bounds what a crash may erase from detailed vote history. +// Intended guarantee: losing recent history must not authorize conflicting +// voting/leader actions after restart. It does NOT guarantee that every vote +// survives on disk, immediate restart voting, or recovery from safety-file rollback. +// +// On an enrolled restart, H is the startup reservation. Without a clean-history +// seal, all vote types (including restored votes) are forbidden at slots <= H; +// slots > H also wait until verified finality/checkpoint state reaches H. +// Leaders always obey that startup barrier, even after a clean vote-history seal. +// The current acknowledged Through separately caps every new signing permission. +// Renewal can raise Through, but never moves this run's fixed recovery barrier. +// +// The worker owns record after startup. Only a successful file+directory sync +// publishes Through to signers; a request, queued write or uncertain sync cannot. +// This assumes storage honors sync and a single fenced identity owner preserves +// the current reservation independently of AccountsDB. See docs/reserved-vote-history.md. +type signingReservation struct { + uncertain bool // Worker only, read after halt. Failed sync may have reached storage. + record alpenglow.VoteReservation + through atomic.Uint64 + desired atomic.Uint64 + stopped atomic.Bool + recoverThrough uint64 // Startup H, or zero after first enrollment / a validated clean-history seal. + leaderThrough uint64 // Exact leader production history is not saved: always skip the old range. + wake chan struct{} + changed chan struct{} + stop chan struct{} + done chan struct{} + stopOnce sync.Once + persist func(alpenglow.VoteReservation) error +} + +func openSigningReservation(cfg VotingConfig, node solana.PublicKey, shredVersion uint16, history *alpenglow.VoteHistory) (*signingReservation, error) { + if cfg.Genesis == (solana.Hash{}) { + return nil, errors.New("reserved voting requires the bound genesis hash") + } + expected := alpenglow.VoteReservation{Version: 1, Node: node, VoteAccount: cfg.VoteAccount, AuthorizedVoter: solana.PublicKey(cfg.AuthorizedVoter.Public().(ed25519.PublicKey)), Genesis: cfg.Genesis, ShredVersion: shredVersion} + record, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, node) + initializing := errors.Is(err, os.ErrNotExist) + if initializing { + if !cfg.InitializeVoteReservation || history.ReservationRequired { + return nil, errors.New("missing vote reservation; explicit first enrollment with complete synchronous history is required") + } + record = expected + record.Generation = 1 + record.Through = history.Root + for slot := range history.VotesCast { + record.Through = max(record.Through, slot) + } + // The baseline is made durable before the first reservation is created. + if err := alpenglow.SaveVoteHistory(cfg.HistoryDir, history, cfg.Identity); err != nil { + return nil, err + } + } else if err != nil { + return nil, fmt.Errorf("refuse unsafe reservation reset: %w", err) + } + if record.Node != expected.Node || record.VoteAccount != expected.VoteAccount || record.AuthorizedVoter != expected.AuthorizedVoter || record.Genesis != expected.Genesis || record.ShredVersion != expected.ShredVersion { + return nil, errors.New("vote reservation cluster or signing identity mismatch; explicit domain migration is required") + } + if record.Through == math.MaxUint64 || record.Generation == math.MaxUint64 { + return nil, errors.New("vote reservation exhausted") + } + for slot := range history.VotesCast { + if slot > record.Through { + return nil, fmt.Errorf("history slot %d exceeds durable reservation %d", slot, record.Through) + } + } + r := &signingReservation{record: record, recoverThrough: record.Through, leaderThrough: record.Through, wake: make(chan struct{}, 1), changed: make(chan struct{}, 1), stop: make(chan struct{}), done: make(chan struct{})} + r.persist = func(next alpenglow.VoteReservation) error { + return alpenglow.SaveVoteReservation(cfg.HistoryDir, next, cfg.Identity) + } + if initializing { + r.recoverThrough = 0 + } else if len(record.CleanHistoryDigest) != 0 { + digest, err := alpenglow.VoteHistoryDigest(cfg.HistoryDir, node) + if err != nil { + return nil, err + } + if bytes.Equal(digest, record.CleanHistoryDigest) { + r.recoverThrough = 0 + } + } + // A matching digest proves exact history only for the sealed session. Consume + // that exception with a durably acknowledged dirty successor before allowing + // new vote/leader signing or new detailed-history decisions. A crash after + // this write must use H, + // even if the detailed history file still looks valid or matches the old seal. + r.record.CleanHistoryDigest = nil + r.record.Generation++ + if err := r.persist(r.record); err != nil { + return nil, fmt.Errorf("consume vote reservation session: %w", err) + } + history.ReservationRequired = true + // Version 2 is deliberately rejected by older binaries that do not enforce H. + if err := alpenglow.SaveVoteHistory(cfg.HistoryDir, history, cfg.Identity); err != nil { + return nil, err + } + r.through.Store(r.record.Through) + mlog.Log.Infof("ALPENGLOW signing reservation: through=%d recovery_through=%d leader_recovery_through=%d", r.record.Through, r.recoverThrough, r.leaderThrough) + go r.run() + return r, nil +} + +// allow is nonblocking and protects vote signing/restoration and leader slots. +// For a nonzero startup barrier H, require slot > H AND finalized >= H. Merely +// observing a new block, waiting elapsed time, or replaying past H is not proof +// that prior decisions can be forgotten. Finality comes from verified consensus/ +// checkpoint state, never an RPC tip; --wait-to-vote-slot cannot override it. +// Passing this gate is necessary, not sufficient: normal protocol checks apply. +func (r *signingReservation) allow(slot, finalized uint64, leader bool) bool { + if r.stopped.Load() { + return false + } + floor := r.recoverThrough + if leader { + floor = r.leaderThrough + } + if floor != 0 && (slot <= floor || finalized < floor) { + return false + } + r.request(slot) + return slot <= r.through.Load() +} + +func (r *signingReservation) request(slot uint64) { + for { + old := r.desired.Load() + if slot <= old || r.desired.CompareAndSwap(old, slot) { + break + } + } + through := r.through.Load() + if slot > through || through-slot <= signingRenewRemaining { + select { + case r.wake <- struct{}{}: + default: + } + } +} + +func (r *signingReservation) run() { + defer close(r.done) + retry := time.NewTicker(250 * time.Millisecond) + defer retry.Stop() + var warned bool + for { + select { + case <-r.stop: + return + case <-r.wake: + case <-retry.C: + } + if r.stopped.Load() { + return + } + slot := r.desired.Load() + if slot == 0 || (slot <= r.record.Through && r.record.Through-slot > signingRenewRemaining) { + continue + } + if slot > math.MaxUint64-signingReserveSlots || r.record.Generation == math.MaxUint64 { + continue + } // Never wrap or grant permission. + next := r.record + next.Through = max(next.Through, slot+signingReserveSlots) + next.Generation++ + next.CleanHistoryDigest = nil + if err := r.persist(next); err != nil { + r.uncertain = true + if !warned { + mlog.Log.Errorf("ALPENGLOW signing reservation renewal failed; permission remains through %d: %v", r.through.Load(), err) + warned = true + } + continue // Retry the same or a greater bound; an uncertain sync never grants permission. + } + warned = false + r.uncertain = false + r.record = next + r.through.Store(next.Through) + select { + case r.changed <- struct{}{}: + default: + } + } +} + +func (r *signingReservation) halt() { + r.stopOnce.Do(func() { r.stopped.Store(true); close(r.stop) }) + <-r.done +} + +// seal may be called only after the voter loop and leader producer have stopped, +// and after the ordered history writer has been drained/joined. The caller must +// also establish verified finality >= recoverThrough and no latched safety fault. +// Sync exact history first, then sync its digest in the reservation. A normal +// process exit or successful unsynced rename alone is not a clean seal. On an +// error the caller must not assume cleanliness; restart validates whichever +// durable record survived. The seal never relaxes the next run's leader barrier. +func (r *signingReservation) seal(dir string, history *alpenglow.VoteHistory, identity ed25519.PrivateKey) error { + r.halt() + if r.uncertain { + return errors.New("uncertain reservation write; retaining unclean recovery") + } + if err := alpenglow.SaveVoteHistory(dir, history, identity); err != nil { + return err + } + digest, err := alpenglow.VoteHistoryDigest(dir, history.NodePubkey) + if err != nil { + return err + } + if r.record.Generation == math.MaxUint64 { + return errors.New("vote reservation generation exhausted") + } + next := r.record + next.Generation++ + next.CleanHistoryDigest = digest + return r.persist(next) +} + +// Retain only events blocked on renewal, not historical catch-up traffic. +// Replay the original event after acknowledgement so normal finality, parent, +// execution and invalidation checks still decide whether to vote. +func (v *alpenglowVoter) retainReservationEvent(event voterEvent) { + r := v.reservation + if r == nil { + return + } + var slot uint64 + switch event.kind { + case voterEventBlock: + slot = event.block.Block.Slot + case voterEventBlockTimeout, voterEventCrashedLeaderTimeout: + slot = event.slot | (alpenglow.LeaderWindowSlots - 1) + case voterEventConsensus: + switch event.consensus.Kind { + case alpenglow.ConsensusEventBlockNotarized, alpenglow.ConsensusEventParentReady, alpenglow.ConsensusEventSafeToNotar, alpenglow.ConsensusEventSafeToSkip: + slot = event.consensus.Slot + if event.consensus.Kind == alpenglow.ConsensusEventSafeToNotar || event.consensus.Kind == alpenglow.ConsensusEventSafeToSkip { + slot |= alpenglow.LeaderWindowSlots - 1 + } + default: + return + } + default: + return + } + if slot <= r.through.Load() || slot <= v.admissionFloor() || slot < v.waitToVoteSlot || v.engine.alpenglowVerifiedFinalityFloor() < r.recoverThrough || slot <= r.recoverThrough { + return + } + if !v.votingStarted && v.readyToVote != nil && !v.readyToVote(slot) { + return + } + r.request(slot) + if len(v.reservationEvents) < votorEventQueueSize { + v.reservationEvents = append(v.reservationEvents, event) + } +} + +// AlpenglowCanSignLeaderSlot protects every produced slot, including the +// trailing slots of a leader window. A clean vote-history marker is not a +// complete leader-block history, so leaders always skip the old reservation. +func (e *AlpenglowObserverEngine) AlpenglowCanSignLeaderSlot(slot uint64) bool { + if e.safetyError() != nil { + return false + } + e.voterMu.RLock() + defer e.voterMu.RUnlock() + v := e.voter + if v == nil { + return false + } + if v.reservation == nil { + return true + } + return v.reservation.allow(slot, e.alpenglowVerifiedFinalityFloor(), true) +} diff --git a/pkg/consensus/vote_reservation_test.go b/pkg/consensus/vote_reservation_test.go new file mode 100644 index 000000000..f4e5c418d --- /dev/null +++ b/pkg/consensus/vote_reservation_test.go @@ -0,0 +1,330 @@ +package consensus + +import ( + "crypto/ed25519" + "errors" + "math" + "os" + "sync/atomic" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func reservedTestConfig(dir string) VotingConfig { + return VotingConfig{Identity: voterTestKey(11), AuthorizedVoter: voterTestKey(12), VoteAccount: solana.PublicKey(voterTestKey(13).Public().(ed25519.PublicKey)), HistoryDir: dir, Genesis: solana.Hash{1}, ReservedHistory: true, InitializeVoteReservation: true, WaitToVoteSlot: 40, ReadyToVote: func(uint64) bool { return true }, EpochForSlot: func(uint64) uint64 { return 7 }, Peers: func([]alpenglow.ValidatorStake) []alpenglow.VotorPeer { return nil }} +} + +func openReservedTestVoter(t *testing.T, cfg VotingConfig, root uint64) (*alpenglowVoter, error) { + t.Helper() + e, err := NewEngine(Config{AlpenglowIdentity: cfg.Identity, AlpenglowShredVersion: 0x1234}) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, e.Close()) }) + set := voterTestValidatorSet(t, cfg.Identity, cfg.AuthorizedVoter, cfg.VoteAccount) + e.SetAlpenglowEpochLookup(cfg.EpochForSlot) + require.NoError(t, e.SetAlpenglowValidatorSet(set)) + block := alpenglow.BlockID{Slot: root, Hash: solana.Hash{byte(root)}} + e.SetAlpenglowRoot(block) + v, err := newAlpenglowVoterUnstarted(e, cfg, block, []alpenglow.ValidatorSet{set}) + if err == nil { + t.Cleanup(func() { require.NoError(t, v.close()) }) + } + return v, err +} + +func reserveThrough(t *testing.T, r *signingReservation, slot uint64) { + t.Helper() + r.request(slot) + require.Eventually(t, func() bool { return r.through.Load() >= slot }, time.Second, time.Millisecond) +} + +// Simulate loss of this process without executing the clean shutdown protocol. +func crashReservedTestVoter(t *testing.T, v *alpenglowVoter) { + t.Helper() + v.shutdownOnce.Do(func() { + v.closeOnce.Do(func() { close(v.done) }) + v.wg.Wait() + v.reservation.halt() + if v.historyWriter != nil { + require.NoError(t, v.historyWriter.close()) + } + require.NoError(t, v.broadcaster.Close()) + require.NoError(t, v.historyLock.Close()) + }) +} + +func TestReservedVotingLostHistorySuffixAndRepeatedCrash(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 60) + baseline, err := os.ReadFile(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node)) + require.NoError(t, err) + voted, err := v.cast(alpenglow.NewNotarizationVote(60, solana.Hash{1}), false) + require.NoError(t, err) + require.True(t, voted) + oldH := v.reservation.through.Load() + crashReservedTestVoter(t, v) + // Reproduce a host crash retaining a valid older version of detailed history. + require.NoError(t, os.WriteFile(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node), baseline, 0600)) + cfg.InitializeVoteReservation = false + resumed, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.Equal(t, oldH, resumed.reservation.recoverThrough) + for _, vote := range []alpenglow.Vote{alpenglow.NewNotarizationVote(60, solana.Hash{2}), alpenglow.NewSkipVote(60), alpenglow.NewFinalizationVote(60), alpenglow.NewNotarizationFallbackVote(60, solana.Hash{2}), alpenglow.NewSkipFallbackVote(60), alpenglow.NewSkipVote(oldH + 1)} { + // Check at the signing boundary, including the restoration bypass. + for _, normal := range []bool{false, true} { + _, _, err := resumed.sign(vote, normal) + require.ErrorIs(t, err, errVoterNotReady) + } + } + require.Equal(t, oldH, resumed.reservation.through.Load(), "recovery must not keep moving its target") + crashReservedTestVoter(t, resumed) + resumed, err = openReservedTestVoter(t, cfg, oldH) + require.NoError(t, err) + require.Equal(t, oldH, resumed.reservation.recoverThrough) + reserveThrough(t, resumed.reservation, oldH+1) + voted, err = resumed.cast(alpenglow.NewSkipVote(oldH+1), false) + require.NoError(t, err) + require.True(t, voted) + voted, err = resumed.cast(alpenglow.NewSkipVote(oldH), false) + require.NoError(t, err) + require.False(t, voted) +} + +func TestReservedVotingCleanMarkerConsumedBeforeSigning(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + voted, err := v.cast(alpenglow.NewSkipVote(44), false) + require.NoError(t, err) + require.True(t, voted) + require.NoError(t, v.close()) + r, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.NotEmpty(t, r.CleanHistoryDigest) + cfg.InitializeVoteReservation = false + resumed, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.Zero(t, resumed.reservation.recoverThrough) + require.True(t, resumed.history.HasSkipped(44)) + r, err = alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Empty(t, r.CleanHistoryDigest) + voted, err = resumed.cast(alpenglow.NewSkipVote(45), false) + require.NoError(t, err) + require.True(t, voted) + require.False(t, resumed.reservation.allow(45, 39, true), "clean vote history does not authorize repeating leader blocks") + crashReservedTestVoter(t, resumed) + again, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.Equal(t, r.Through, again.reservation.recoverThrough) + // A shutdown before recovering the uncertain range must not mark it clean. + require.NoError(t, again.close()) + r, err = alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Empty(t, r.CleanHistoryDigest) +} + +func TestReservedVotingRejectsMissingCorruptOrWrongDomain(t *testing.T) { + for _, which := range []string{"missing_history", "missing_bound", "corrupt_bound", "genesis", "authorized", "vote_account", "synchronous"} { + t.Run(which, func(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + crashReservedTestVoter(t, v) + cfg.InitializeVoteReservation = false + switch which { + case "missing_history": + require.NoError(t, os.Remove(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node))) + case "missing_bound": + require.NoError(t, os.Remove(alpenglow.VoteReservationFilename(cfg.HistoryDir, v.node))) + cfg.InitializeVoteReservation = true + case "corrupt_bound": + require.NoError(t, os.WriteFile(alpenglow.VoteReservationFilename(cfg.HistoryDir, v.node), []byte("{"), 0600)) + case "genesis": + cfg.Genesis = solana.Hash{2} + case "authorized": + cfg.AuthorizedVoter = voterTestKey(19) + case "vote_account": + cfg.VoteAccount = solana.PublicKey{9} + case "synchronous": + cfg.ReservedHistory = false + } + _, err = openReservedTestVoter(t, cfg, 39) + require.Error(t, err) + }) + } +} + +func TestReservedVotingRequiresEnrollmentAndExclusiveOwner(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + cfg.InitializeVoteReservation = false + _, err := openReservedTestVoter(t, cfg, 39) + require.Error(t, err) + cfg.InitializeVoteReservation = true + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + _, err = openReservedTestVoter(t, cfg, 39) + require.ErrorContains(t, err, "already owned") + require.NoError(t, v.close()) +} + +func TestSigningReservationUnacknowledgedSyncCannotAuthorize(t *testing.T) { + entered, release := make(chan struct{}), make(chan struct{}) + r := &signingReservation{record: alpenglow.VoteReservation{Through: 64, Generation: 1}, wake: make(chan struct{}, 1), changed: make(chan struct{}, 1), stop: make(chan struct{}), done: make(chan struct{})} + r.through.Store(64) + var calls atomic.Uint64 + r.persist = func(next alpenglow.VoteReservation) error { + calls.Add(1) + close(entered) + <-release + return nil + } + go r.run() + require.True(t, r.allow(64, 64, false)) + <-entered + require.Equal(t, uint64(64), r.through.Load()) + require.False(t, r.allow(65, 64, false)) + close(release) + require.Eventually(t, func() bool { return r.through.Load() > 64 }, time.Second, time.Millisecond) + require.True(t, r.allow(65, 64, false)) + r.halt() + require.Equal(t, uint64(1), calls.Load()) +} + +func TestSigningReservationUncertainWriteSurvivesRestart(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + // Stop the original worker, then exercise a fresh worker against the real record. + v.reservation.halt() + record, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + r := &signingReservation{record: record, wake: make(chan struct{}, 1), changed: make(chan struct{}, 1), stop: make(chan struct{}), done: make(chan struct{})} + r.through.Store(record.Through) + wrote := make(chan struct{}, 1) + r.persist = func(next alpenglow.VoteReservation) error { + err := alpenglow.SaveVoteReservation(cfg.HistoryDir, next, cfg.Identity) + select { + case wrote <- struct{}{}: + default: + } + if err != nil { + return err + } + return errors.New("injected lost sync acknowledgement") + } + v.reservation = r + go r.run() + require.False(t, r.allow(60, 39, false)) + <-wrote + r.halt() + require.Equal(t, record.Through, r.through.Load()) + durable, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Greater(t, durable.Through, record.Through) + require.ErrorContains(t, r.seal(cfg.HistoryDir, v.history, cfg.Identity), "uncertain") + crashReservedTestVoter(t, v) + cfg.InitializeVoteReservation = false + resumed, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.Equal(t, durable.Through, resumed.reservation.recoverThrough) +} + +func TestSigningReservationNeverWraps(t *testing.T) { + r := &signingReservation{record: alpenglow.VoteReservation{Through: 64, Generation: 1}, wake: make(chan struct{}, 1), changed: make(chan struct{}, 1), stop: make(chan struct{}), done: make(chan struct{}), persist: func(alpenglow.VoteReservation) error { t.Error("overflow attempted persistence"); return nil }} + r.through.Store(64) + go r.run() + require.False(t, r.allow(math.MaxUint64, 64, false)) + r.halt() + require.Equal(t, uint64(64), r.through.Load()) +} + +func TestReservationRetryRechecksFinality(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + event := voterEvent{kind: voterEventBlockTimeout, slot: 44} + require.NoError(t, v.handle(event)) + require.NotEmpty(t, v.reservationEvents) + reserveThrough(t, v.reservation, 44) + v.engine.SetAlpenglowRoot(alpenglow.BlockID{Slot: 47, Hash: solana.Hash{47}}) + pending := v.reservationEvents + v.reservationEvents = nil + for _, e := range pending { + require.NoError(t, v.handle(e)) + } + require.False(t, v.history.HasSkipped(44), "finalized work must not be signed after a delayed ack") +} + +func TestReservedVotingEverySignatureTypeAtBound(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + h := v.reservation.through.Load() + // Freeze acknowledgement while leaving the signing guard active. + v.reservation.halt() + v.reservation.stopped.Store(false) + defer v.reservation.stopped.Store(true) + for _, slot := range []uint64{h, h + 1} { + votes := []alpenglow.Vote{alpenglow.NewNotarizationVote(slot, solana.Hash{2}), alpenglow.NewSkipVote(slot), alpenglow.NewFinalizationVote(slot), alpenglow.NewNotarizationFallbackVote(slot, solana.Hash{2}), alpenglow.NewSkipFallbackVote(slot)} + for _, vote := range votes { + for _, normal := range []bool{false, true} { + _, _, err := v.sign(vote, normal) + if slot == h { + require.NoError(t, err) + } else { + require.ErrorIs(t, err, errVoterNotReady) + } + } + } + } +} + +func TestReservedCleanDigestMismatchUsesCrashRecovery(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + baseline, err := os.ReadFile(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node)) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + voted, err := v.cast(alpenglow.NewSkipVote(44), false) + require.NoError(t, err) + require.True(t, voted) + require.NoError(t, v.close()) + require.NoError(t, os.WriteFile(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node), baseline, 0600)) + resumed, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.NotZero(t, resumed.reservation.recoverThrough) +} + +func TestReservationLoopRetriesAfterAcknowledgement(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + v.start() + require.NoError(t, v.enqueue(voterEvent{kind: voterEventBlockTimeout, slot: 44})) + // Read via stats, not mutable voter history, while its loop is running. + require.Eventually(t, func() bool { v.landingMu.RLock(); defer v.landingMu.RUnlock(); return v.stats.VotesCastThisRun == 4 }, time.Second, time.Millisecond) + require.NoError(t, v.close()) + for slot := uint64(44); slot <= 47; slot++ { + require.True(t, v.history.HasSkipped(slot)) + } +} + +func TestReservationRetainsWindowCrossingBound(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + h := v.reservation.through.Load() // 76: window ends at 79. + v.retainReservationEvent(voterEvent{kind: voterEventBlockTimeout, slot: h}) + require.Len(t, v.reservationEvents, 1, "trailing skip slots require retry even if the first slot fits") +} diff --git a/pkg/consensus/voter.go b/pkg/consensus/voter.go index b77a7689b..3f950126b 100644 --- a/pkg/consensus/voter.go +++ b/pkg/consensus/voter.go @@ -5,6 +5,7 @@ import ( "errors" "fmt" "math" + "os" "sort" "sync" "time" @@ -38,43 +39,57 @@ type VotingPeerSource func(validators []alpenglow.ValidatorStake) []alpenglow.Vo // account address; AuthorizedVoter is the Ed25519 signer from which its BLS key // was registered. type VotingConfig struct { - Identity ed25519.PrivateKey - AuthorizedVoter ed25519.PrivateKey - VoteAccount solana.PublicKey - HistoryDir string - EpochForSlot func(slot uint64) uint64 - Peers VotingPeerSource - SlotDuration time.Duration - WaitToVoteSlot uint64 - ReadyToVote func(slot uint64) bool + Identity ed25519.PrivateKey + AuthorizedVoter ed25519.PrivateKey + VoteAccount solana.PublicKey + HistoryDir string + ReservedHistory bool + InitializeVoteReservation bool + Genesis solana.Hash + EpochForSlot func(slot uint64) uint64 + Peers VotingPeerSource + SlotDuration time.Duration + WaitToVoteSlot uint64 // Inclusive minimum for new votes; authenticated history restoration is separate + ReadyToVote func(slot uint64) bool } // VotingStats exposes positive network evidence separately from local casting. // NetworkLandedVotes counts unique persisted votes whose rank appeared in the // exact BLS-verified certificate proof received over Votor QUIC. type VotingStats struct { - Enabled bool `json:"enabled"` - VotesCastThisRun uint64 `json:"votes_cast_this_run"` - NetworkLandedVotes uint64 `json:"network_landed_votes"` - LastNetworkLandedSlot uint64 `json:"last_network_landed_slot,omitempty"` - LastNetworkLandedVoteType alpenglow.VoteType `json:"last_network_landed_vote_type,omitempty"` - LastNetworkCertificateType alpenglow.CertificateType `json:"last_network_certificate_type,omitempty"` - LastNetworkLandedAt time.Time `json:"last_network_landed_at,omitempty"` - BroadcastMessagesQueued uint64 `json:"broadcast_messages_queued"` - BroadcastMessagesDropped uint64 `json:"broadcast_messages_dropped"` - BroadcastPeerSends uint64 `json:"broadcast_peer_sends"` - BroadcastPeerSendsSkipped uint64 `json:"broadcast_peer_sends_skipped"` - BroadcastPeerSendErrors uint64 `json:"broadcast_peer_send_errors"` - BroadcastDesiredPeers int `json:"broadcast_desired_peers"` - BroadcastActiveConnections int `json:"broadcast_active_connections"` - BroadcastPendingConnections int `json:"broadcast_pending_connections"` - BroadcastConnectionAttempts uint64 `json:"broadcast_connection_attempts"` - BroadcastConnectionErrors uint64 `json:"broadcast_connection_errors"` - BroadcastConnectionJobsDropped uint64 `json:"broadcast_connection_jobs_dropped"` - BroadcastLastPeerSendError string `json:"broadcast_last_peer_send_error,omitempty"` - BroadcastLastPeerSendErrorAt time.Time `json:"broadcast_last_peer_send_error_at,omitempty"` - BroadcastLastConnectionError string `json:"broadcast_last_connection_error,omitempty"` - BroadcastLastConnectionErrorAt time.Time `json:"broadcast_last_connection_error_at,omitempty"` + HistorySnapshotsSubmitted uint64 `json:"history_snapshots_submitted,omitempty"` + HistorySnapshotsWritten uint64 `json:"history_snapshots_written,omitempty"` + HistorySnapshotsCoalesced uint64 `json:"history_snapshots_coalesced,omitempty"` + ReservedHistory bool `json:"reserved_history"` + SigningReservedThrough uint64 `json:"signing_reserved_through,omitempty"` + RecoveryThrough uint64 `json:"recovery_through,omitempty"` + Enabled bool `json:"enabled"` + VotesCastThisRun uint64 `json:"votes_cast_this_run"` + NetworkLandedVotes uint64 `json:"network_landed_votes"` + LastNetworkLandedSlot uint64 `json:"last_network_landed_slot,omitempty"` + LastNetworkLandedVoteType alpenglow.VoteType `json:"last_network_landed_vote_type,omitempty"` + LastNetworkCertificateType alpenglow.CertificateType `json:"last_network_certificate_type,omitempty"` + LastNetworkLandedAt time.Time `json:"last_network_landed_at,omitempty"` + BroadcastMessagesQueued uint64 `json:"broadcast_messages_queued"` + BroadcastMessagesDropped uint64 `json:"broadcast_messages_dropped"` + BroadcastPeerSends uint64 `json:"broadcast_peer_sends"` + BroadcastPeerSendsSkipped uint64 `json:"broadcast_peer_sends_skipped"` + BroadcastPeerSendErrors uint64 `json:"broadcast_peer_send_errors"` + BroadcastPeerQueueDrops uint64 `json:"broadcast_peer_queue_drops"` + BroadcastPeerQueueDiscarded uint64 `json:"broadcast_peer_queue_discarded"` + BroadcastPeerSendTimeouts uint64 `json:"broadcast_peer_send_timeouts"` + BroadcastPeerQueueMaxDelay time.Duration `json:"broadcast_peer_queue_max_delay_ns"` + BroadcastPeerQueues []alpenglow.VotorPeerQueueStats `json:"broadcast_peer_queues,omitempty"` + BroadcastDesiredPeers int `json:"broadcast_desired_peers"` + BroadcastActiveConnections int `json:"broadcast_active_connections"` + BroadcastPendingConnections int `json:"broadcast_pending_connections"` + BroadcastConnectionAttempts uint64 `json:"broadcast_connection_attempts"` + BroadcastConnectionErrors uint64 `json:"broadcast_connection_errors"` + BroadcastConnectionJobsDropped uint64 `json:"broadcast_connection_jobs_dropped"` + BroadcastLastPeerSendError string `json:"broadcast_last_peer_send_error,omitempty"` + BroadcastLastPeerSendErrorAt time.Time `json:"broadcast_last_peer_send_error_at,omitempty"` + BroadcastLastConnectionError string `json:"broadcast_last_connection_error,omitempty"` + BroadcastLastConnectionErrorAt time.Time `json:"broadcast_last_connection_error_at,omitempty"` } type voterEventKind uint8 @@ -88,6 +103,7 @@ const ( voterEventValidatorSet voterEventRoot voterEventNetworkCertificate + voterEventDurableRoot ) type voterEvent struct { @@ -109,43 +125,49 @@ type pendingVotorBlock struct { // history decisions run on loop; validator-set snapshots are protected only so // the outbound peer callback can read them from broadcast workers. type alpenglowVoter struct { - engine *AlpenglowObserverEngine - identity ed25519.PrivateKey - node solana.PublicKey - voteAccount solana.PublicKey - signer *alpenglow.BLSSigner - historyDir string - history *alpenglow.VoteHistory - epochForSlot func(uint64) uint64 - peerSource VotingPeerSource - slotDuration time.Duration - waitToVoteSlot uint64 - readyToVote func(slot uint64) bool - broadcaster *alpenglow.VotorBroadcaster - events chan voterEvent - done chan struct{} - startOnce sync.Once - closeOnce sync.Once - wg sync.WaitGroup - setsMu sync.RWMutex - sets map[uint64]alpenglow.ValidatorSet - restored map[alpenglow.VoteMessageKey]bool - pending map[uint64][]pendingVotorBlock - receivedShred map[uint64]bool - timeoutsSet map[uint64]bool - executedBlocks map[alpenglow.BlockID]bool - highestFinal uint64 - lastFinalizedAt time.Time - votingStarted bool - latestLiveSlot uint64 - standstillSlot *uint64 - refreshQueue []alpenglow.Message - refreshCursor int - lastWarn map[uint64]time.Time - landingMu sync.RWMutex - landed map[alpenglow.VoteMessageKey]struct{} - stats VotingStats - lastStatsLog time.Time + engine *AlpenglowObserverEngine + identity ed25519.PrivateKey + node solana.PublicKey + voteAccount solana.PublicKey + signer *alpenglow.BLSSigner + historyDir string + historyLock *os.File + reservation *signingReservation + historyWriter *voteHistoryWriter + reservationEvents []voterEvent + shutdownOnce sync.Once + shutdownErr error + history *alpenglow.VoteHistory + epochForSlot func(uint64) uint64 + peerSource VotingPeerSource + slotDuration time.Duration + waitToVoteSlot uint64 + readyToVote func(slot uint64) bool + broadcaster *alpenglow.VotorBroadcaster + events chan voterEvent + done chan struct{} + startOnce sync.Once + closeOnce sync.Once + wg sync.WaitGroup + setsMu sync.RWMutex + sets map[uint64]alpenglow.ValidatorSet + restored map[alpenglow.VoteMessageKey]bool + pending map[uint64][]pendingVotorBlock + receivedShred map[uint64]bool + timeoutsSet map[uint64]bool + executedBlocks map[alpenglow.BlockID]bool + highestFinal uint64 + lastFinalizedAt time.Time + votingStarted bool + latestLiveSlot uint64 + standstillSlot *uint64 + refreshQueue []alpenglow.Message + refreshCursor int + lastWarn map[uint64]time.Time + landingMu sync.RWMutex + landed map[alpenglow.VoteMessageKey]struct{} + stats VotingStats + lastStatsLog time.Time // beforeVoteGuard is a deterministic test seam for invalidation races. It // is nil in production. beforeVoteGuard func(alpenglow.BlockID) @@ -163,6 +185,9 @@ func newAlpenglowVoterUnstarted(engine *AlpenglowObserverEngine, cfg VotingConfi } func newAlpenglowVoterWithStart(engine *AlpenglowObserverEngine, cfg VotingConfig, root alpenglow.BlockID, start bool, sets []alpenglow.ValidatorSet) (*alpenglowVoter, error) { + if cfg.InitializeVoteReservation && !cfg.ReservedHistory { + return nil, errors.New("initialize-vote-reservation requires reserved-vote-history") + } if engine == nil { return nil, fmt.Errorf("enable Alpenglow voting: nil consensus engine") } @@ -192,17 +217,52 @@ func newAlpenglowVoterWithStart(engine *AlpenglowObserverEngine, cfg VotingConfi if node != engineNode { return nil, fmt.Errorf("enable Alpenglow voting: identity %s does not match consensus transport identity %s", node, engineNode) } + historyLock, err := alpenglow.LockVoteHistory(cfg.HistoryDir, node) + if err != nil { + return nil, err + } + keepLock := false + defer func() { + if !keepLock { + historyLock.Close() + } + }() + if !cfg.ReservedHistory { + if _, err := os.Stat(alpenglow.VoteReservationFilename(cfg.HistoryDir, node)); !errors.Is(err, os.ErrNotExist) { + return nil, fmt.Errorf("existing or unreadable vote reservation requires reserved history mode") + } + } history, err := alpenglow.LoadVoteHistory(cfg.HistoryDir, node) if err != nil { if !errors.Is(err, alpenglow.ErrVoteHistoryNotFound) { return nil, fmt.Errorf("enable Alpenglow voting: refuse unsafe vote-history reset: %w", err) } + if cfg.ReservedHistory { + if _, err := os.Stat(alpenglow.VoteReservationFilename(cfg.HistoryDir, node)); !errors.Is(err, os.ErrNotExist) || !cfg.InitializeVoteReservation { + return nil, fmt.Errorf("missing history for reserved voter; refusing automatic reset") + } + } history = alpenglow.NewVoteHistory(node, root.Slot) if err := alpenglow.SaveVoteHistory(cfg.HistoryDir, history, cfg.Identity); err != nil { return nil, fmt.Errorf("initialize Alpenglow vote history: %w", err) } mlog.Log.FileOnlyf("ALPENGLOW voting: created new vote history for %s at root %d; do not reuse this vote account on another validator", node, root.Slot) } + if history.ReservationRequired && !cfg.ReservedHistory { + return nil, errors.New("reserved vote history cannot be opened in synchronous mode") + } + var reservation *signingReservation + if cfg.ReservedHistory { + reservation, err = openSigningReservation(cfg, node, engine.shredVersion, history) + if err != nil { + return nil, err + } + defer func() { + if !keepLock { + reservation.halt() + } + }() + } if history.Root < root.Slot { history.SetRoot(root.Slot) if err := alpenglow.SaveVoteHistory(cfg.HistoryDir, history, cfg.Identity); err != nil { @@ -220,6 +280,8 @@ func newAlpenglowVoterWithStart(engine *AlpenglowObserverEngine, cfg VotingConfi voteAccount: cfg.VoteAccount, signer: signer, historyDir: cfg.HistoryDir, + historyLock: historyLock, + reservation: reservation, history: history, epochForSlot: cfg.EpochForSlot, peerSource: cfg.Peers, @@ -261,6 +323,12 @@ func newAlpenglowVoterWithStart(engine *AlpenglowObserverEngine, cfg VotingConfi return nil, err } v.broadcaster = broadcaster + if reservation != nil { + v.historyWriter = newVoteHistoryWriter(func(snapshot *alpenglow.VoteHistorySnapshot) error { + return alpenglow.SaveReservedVoteHistorySnapshot(v.historyDir, snapshot) + }, v.failHistoryWrite) + } + keepLock = true if start { v.start() } @@ -300,12 +368,26 @@ func (v *alpenglowVoter) loop() { v.closeOnce.Do(func() { close(v.done) }) v.wg.Done() }() + var reservationChanged <-chan struct{} + if v.reservation != nil { + reservationChanged = v.reservation.changed + } ticker := time.NewTicker(time.Second) defer ticker.Stop() for { select { case <-v.done: return + case <-reservationChanged: + pending := v.reservationEvents + v.reservationEvents = nil + for _, event := range pending { + if err := v.handle(event); err != nil { + v.engine.latchSafetyError(fmt.Errorf("reservation retry: %w", err)) + mlog.Log.Errorf("ALPENGLOW VOTING SAFETY: reservation retry: %v", err) + return + } + } case event := <-v.events: if err := v.handle(event); err != nil { v.engine.latchSafetyError(fmt.Errorf("voting engine: %w", err)) @@ -324,6 +406,11 @@ func (v *alpenglowVoter) loop() { } func (v *alpenglowVoter) handle(event voterEvent) error { + v.retainReservationEvent(event) + return v.handleEvent(event) +} + +func (v *alpenglowVoter) handleEvent(event voterEvent) error { floor := v.admissionFloor() if event.kind == voterEventBlock && v.engine.ensureChain().IsObjectivelyInvalidBlock(event.block.Block) { return nil @@ -363,6 +450,9 @@ func (v *alpenglowVoter) handle(event voterEvent) error { return nil } return v.saveHistory() + case voterEventDurableRoot: + root := v.engine.applyAlpenglowDurableRoot(event.slot) + return v.handleEvent(voterEvent{kind: voterEventRoot, root: root}) case voterEventNetworkCertificate: v.recordNetworkCertificate(event.certificate) return nil @@ -477,9 +567,9 @@ func (v *alpenglowVoter) handleConsensus(event alpenglow.ConsensusEvent) error { func (v *alpenglowVoter) admissionFloor() uint64 { floor := v.history.Root - if v.highestFinal > floor { - floor = v.highestFinal - } + // highestFinal tracks network progress and standstill, not retirement of + // our own decisions. Keep ParentReady, pending replay and exact vote history + // available until the retained pool or an ordered durable root retires them. if engineFloor := v.engine.alpenglowVoteActionFloor(); engineFloor > floor { floor = engineFloor } @@ -661,7 +751,7 @@ func (v *alpenglowVoter) castTarget(vote alpenglow.Vote, restoring bool, guarded if !restoring && v.beforeVoteGuard != nil { v.beforeVoteGuard(guardedBlock) } - // Hold through signing, durable history, pool admission, and broadcast. + // Hold through signing, history recording, pool admission, and broadcast. // Objective invalidation takes the write side before changing the chain, // so a new or restored vote is wholly before it or sees the tombstone. v.engine.invalidActionMu.RLock() @@ -682,18 +772,22 @@ func (v *alpenglowVoter) castTarget(vote alpenglow.Vote, restoring bool, guarded return false, nil } if !restoring { - // Finality can advance while the BLS signature is computed. Avoid a - // durable stale record when that race is already visible here; if it - // advances later, atomic admission below classifies it benignly. + // Retention/root pruning can advance while the BLS signature is computed. + // Avoid an expired record if that race is already visible here; atomic + // admission below classifies a later pruning race benignly. if vote.Slot <= v.admissionFloor() { return false, nil } if err := v.history.AddVote(vote); err != nil { return false, fmt.Errorf("record %s vote at slot %d: %w", vote.Type, vote.Slot, err) } - // Pool admission may synchronously assemble and publish a certificate. - // Persist the anti-equivocation record first so no externally visible - // proof can survive a crash without its signed local history. + // Pool admission can publish a certificate. In reserved mode the durable + // upper bound covers loss of this unsynchronized history replacement; + // synchronous mode still persists the exact history before admission. + // sign computed BLS bytes in RAM, but nothing may expose them before + // this boundary succeeds. In reserved mode saveHistory only queues the + // snapshot: restart safety comes from the durable reservation checked + // before sign, not from assuming this snapshot reached durable storage. if err := v.saveHistory(); err != nil { return false, err } @@ -727,7 +821,16 @@ func (v *alpenglowVoter) castTarget(vote alpenglow.Vote, restoring bool, guarded return true, nil } +// sign checks reservation recovery even when restoration bypasses the live +// joining gate. Re-signing a saved vote is still signing; the presence of an +// older valid history file cannot prove that its lost suffix was conflict-free. func (v *alpenglowVoter) sign(vote alpenglow.Vote, respectVotingGate bool) (alpenglow.VoteMessage, alpenglow.VoteVerifyResult, error) { + if err := v.engine.safetyError(); err != nil { + return alpenglow.VoteMessage{}, alpenglow.VoteVerifyResult{}, err + } + if !respectVotingGate && v.reservation != nil && !v.reservation.allow(vote.Slot, v.engine.alpenglowVerifiedFinalityFloor(), false) { + return alpenglow.VoteMessage{}, alpenglow.VoteVerifyResult{}, fmt.Errorf("%w: waiting for verified recovery or durable signing reservation", errVoterNotReady) + } if respectVotingGate { if err := v.votingGateError(vote.Slot); err != nil { return alpenglow.VoteMessage{}, alpenglow.VoteVerifyResult{}, err @@ -768,7 +871,7 @@ func (v *alpenglowVoter) votingGateError(slot uint64) error { return fmt.Errorf("%w: slot %d is at or below consensus action floor %d", errVoterNotReady, slot, floor) } if slot < v.waitToVoteSlot { - return fmt.Errorf("%w: waiting for startup watermark slot %d", errVoterNotReady, v.waitToVoteSlot) + return fmt.Errorf("%w: waiting for voting cutoff slot %d", errVoterNotReady, v.waitToVoteSlot) } // ReadyToVote is a startup join guard, not a perpetual clock check. Once an // accepted live block or vote joins Votor, verified ParentReady and timeout @@ -777,6 +880,9 @@ func (v *alpenglowVoter) votingGateError(slot uint64) error { if !v.votingStarted && v.readyToVote != nil && !v.readyToVote(slot) { return fmt.Errorf("%w: slot is still behind the startup live voting window", errVoterNotReady) } + if v.reservation != nil && !v.reservation.allow(slot, v.engine.alpenglowVerifiedFinalityFloor(), false) { + return fmt.Errorf("%w: waiting for verified recovery or durable signing reservation", errVoterNotReady) + } return nil } @@ -837,6 +943,9 @@ func (v *alpenglowVoter) votorTransportValidators() []alpenglow.ValidatorStake { } func (v *alpenglowVoter) restoreVotesForEpoch(epoch uint64) error { + if v.reservation != nil && v.reservation.recoverThrough != 0 { + return nil + } floor := v.admissionFloor() for _, vote := range v.history.VotesAfter(v.history.Root - minU64(v.history.Root, 1)) { // Keep every signed history entry for anti-equivocation, but do not @@ -884,6 +993,9 @@ func (v *alpenglowVoter) restoreVotesForEpoch(epoch uint64) error { // persist-before-admission crash window without trusting process-local invalid // block state across a restart. func (v *alpenglowVoter) restoreVotesForBlock(block alpenglow.BlockID) (bool, error) { + if v.reservation != nil && v.reservation.recoverThrough != 0 { + return false, nil + } if block.Slot <= v.admissionFloor() { return false, nil } @@ -1086,13 +1198,29 @@ func (v *alpenglowVoter) isIdentityStaked(slot uint64) bool { return false } +// saveHistory is a publication boundary with different persistence semantics: +// synchronous mode acknowledges durable exact history; reserved mode acknowledges +// an immutable queued snapshot only. The latter relies on the signing reservation. func (v *alpenglowVoter) saveHistory() error { + if v.historyWriter != nil { + snapshot, err := alpenglow.PrepareReservedVoteHistory(v.history, v.identity) + if err != nil { + return err + } + return v.historyWriter.submit(snapshot) + } if err := alpenglow.SaveVoteHistory(v.historyDir, v.history, v.identity); err != nil { return fmt.Errorf("persist vote history before consensus publication: %w", err) } return nil } +func (v *alpenglowVoter) failHistoryWrite(err error) { + v.engine.latchSafetyError(err) + mlog.Log.Errorf("ALPENGLOW VOTING SAFETY: %v", err) + v.closeOnce.Do(func() { close(v.done) }) +} + func (v *alpenglowVoter) recordNetworkCertificate(cert alpenglow.Certificate) { set, ok := v.validatorSet(cert.Slot) if !ok { @@ -1201,6 +1329,14 @@ func (v *alpenglowVoter) snapshot() VotingStats { stats := v.stats v.landingMu.RUnlock() stats.Enabled = true + if v.historyWriter != nil { + stats.HistorySnapshotsSubmitted, stats.HistorySnapshotsWritten, stats.HistorySnapshotsCoalesced = v.historyWriter.counters() + } + if v.reservation != nil { + stats.ReservedHistory = true + stats.SigningReservedThrough = v.reservation.through.Load() + stats.RecoveryThrough = v.reservation.recoverThrough + } if v.broadcaster != nil { broadcast := v.broadcaster.Stats() stats.BroadcastMessagesQueued = broadcast.MessagesQueued @@ -1208,6 +1344,11 @@ func (v *alpenglowVoter) snapshot() VotingStats { stats.BroadcastPeerSends = broadcast.PeerSends stats.BroadcastPeerSendsSkipped = broadcast.PeerSendsSkipped stats.BroadcastPeerSendErrors = broadcast.PeerSendErrors + stats.BroadcastPeerQueueDrops = broadcast.PeerQueueDrops + stats.BroadcastPeerQueueDiscarded = broadcast.PeerQueueDiscarded + stats.BroadcastPeerSendTimeouts = broadcast.PeerSendTimeouts + stats.BroadcastPeerQueueMaxDelay = broadcast.PeerQueueMaxDelay + stats.BroadcastPeerQueues = broadcast.PeerQueues stats.BroadcastDesiredPeers = broadcast.DesiredPeers stats.BroadcastActiveConnections = broadcast.Connections stats.BroadcastPendingConnections = broadcast.PendingConnections @@ -1228,7 +1369,7 @@ func (v *alpenglowVoter) maybeLogStats() { } v.lastStatsLog = time.Now() stats := v.snapshot() - mlog.Log.FileOnlyf("alpenglow voting stats: votes_cast_this_run=%d network_landed=%d last_landed_slot=%d broadcast_queued=%d broadcast_dropped=%d peer_sends=%d peer_sends_skipped=%d peer_send_errors=%d desired_peers=%d active_connections=%d pending_connections=%d connection_attempts=%d connection_errors=%d connection_jobs_dropped=%d", + mlog.Log.FileOnlyf("alpenglow voting stats: votes_cast_this_run=%d network_landed=%d last_landed_slot=%d broadcast_queued=%d broadcast_dropped=%d peer_sends=%d peer_sends_skipped=%d peer_send_errors=%d peer_queue_drops=%d peer_queue_discarded=%d peer_send_timeouts=%d peer_queue_max_delay=%s desired_peers=%d active_connections=%d pending_connections=%d connection_attempts=%d connection_errors=%d connection_jobs_dropped=%d reserved_history=%t signing_through=%d recovery_through=%d history_submitted=%d history_written=%d history_coalesced=%d", stats.VotesCastThisRun, stats.NetworkLandedVotes, stats.LastNetworkLandedSlot, @@ -1237,12 +1378,18 @@ func (v *alpenglowVoter) maybeLogStats() { stats.BroadcastPeerSends, stats.BroadcastPeerSendsSkipped, stats.BroadcastPeerSendErrors, + stats.BroadcastPeerQueueDrops, + stats.BroadcastPeerQueueDiscarded, + stats.BroadcastPeerSendTimeouts, + stats.BroadcastPeerQueueMaxDelay, stats.BroadcastDesiredPeers, stats.BroadcastActiveConnections, stats.BroadcastPendingConnections, stats.BroadcastConnectionAttempts, stats.BroadcastConnectionErrors, stats.BroadcastConnectionJobsDropped, + stats.ReservedHistory, stats.SigningReservedThrough, stats.RecoveryThrough, + stats.HistorySnapshotsSubmitted, stats.HistorySnapshotsWritten, stats.HistorySnapshotsCoalesced, ) } @@ -1274,9 +1421,28 @@ func (v *alpenglowVoter) close() error { if v == nil { return nil } - v.closeOnce.Do(func() { close(v.done) }) - v.wg.Wait() - return v.broadcaster.Close() + v.shutdownOnce.Do(func() { + v.closeOnce.Do(func() { close(v.done) }) + v.wg.Wait() + if v.reservation != nil { + v.reservation.halt() + if v.historyWriter != nil { + v.shutdownErr = v.historyWriter.close() + } + // Exiting normally is insufficient: an unresolved recovery barrier, + // writer failure or safety fault must leave the session unsealed. + floor := v.engine.alpenglowVerifiedFinalityFloor() + if v.shutdownErr == nil && v.engine.safetyError() == nil && floor >= v.reservation.recoverThrough { + v.history.SetRoot(floor) + v.shutdownErr = v.reservation.seal(v.historyDir, v.history, v.identity) + if v.shutdownErr == nil { + mlog.Log.Infof("ALPENGLOW signing reservation: clean history sealed at root=%d through=%d", v.history.Root, v.reservation.through.Load()) + } + } + } + v.shutdownErr = errors.Join(v.shutdownErr, v.broadcaster.Close(), v.historyLock.Close()) + }) + return v.shutdownErr } func pendingContains(blocks []pendingVotorBlock, candidate pendingVotorBlock) bool { diff --git a/pkg/consensus/voter_finality_ordering_test.go b/pkg/consensus/voter_finality_ordering_test.go new file mode 100644 index 000000000..a53abb640 --- /dev/null +++ b/pkg/consensus/voter_finality_ordering_test.go @@ -0,0 +1,192 @@ +package consensus + +import ( + "context" + "crypto/ed25519" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Keep the actor unstarted so tests can place finality, replay and durable +// promotion in a deterministic order, using the real engine event queue. +func newOrderingTestVoter(t *testing.T, reserved bool) (*AlpenglowObserverEngine, *alpenglowVoter) { + t.Helper() + cfg := reservedTestConfig(t.TempDir()) + cfg.ReservedHistory = reserved + cfg.InitializeVoteReservation = reserved + root := alpenglow.BlockID{Slot: 39, Hash: solana.Hash{39}} + e, err := NewEngine(Config{AlpenglowIdentity: cfg.Identity, AlpenglowShredVersion: 0x1234}) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, e.Close()) }) + set := voterTestValidatorSet(t, cfg.Identity, cfg.AuthorizedVoter, cfg.VoteAccount) + e.SetAlpenglowEpochLookup(cfg.EpochForSlot) + require.NoError(t, e.SetAlpenglowValidatorSet(set)) + e.SetAlpenglowRoot(root) + v, err := newAlpenglowVoterUnstarted(e, cfg, root, []alpenglow.ValidatorSet{set}) + require.NoError(t, err) + e.voter = v + if reserved { + reserveThrough(t, v.reservation, 44) + } + // Equivalent to startup's trusted-root ParentReady seed. + require.True(t, v.history.AddParentReady(40, root)) + return e, v +} + +func drainOrderingEvents(t *testing.T, v *alpenglowVoter) { + t.Helper() + for i := 0; i < 1000; i++ { + select { + case event := <-v.events: + require.NoError(t, v.handle(event)) + default: + return + } + } + t.Fatal("voter event queue did not drain") +} + +func observeOrderingBlock(t *testing.T, e *AlpenglowObserverEngine, slot uint64) alpenglow.BlockID { + t.Helper() + id := alpenglow.BlockID{Slot: slot, Hash: solana.Hash{byte(slot)}} + require.NoError(t, e.ObserveBlock(context.Background(), BlockObservation{Source: "ordering-test", Block: &block.Block{ + Slot: slot, ParentSlot: slot - 1, + AlpenglowBlockID: [32]byte(id.Hash), HasAlpenglowBlockID: true, + AlpenglowParentBlockID: [32]byte{byte(slot - 1)}, HasAlpenglowParentBlockID: true, + }})) + return id +} + +func finalizeOrderingBlock(t *testing.T, e *AlpenglowObserverEngine, v *alpenglowVoter, id alpenglow.BlockID) { + t.Helper() + // The two peers supply the 60% slow-finality quorum without our vote; + // adding our 30% notarization later can produce a fast certificate. + for _, vote := range []alpenglow.Vote{alpenglow.NewNotarizationVote(id.Slot, id.Hash), alpenglow.NewFinalizationVote(id.Slot)} { + for rank, key := range []ed25519.PrivateKey{voterTestKey(21), voterTestKey(22)} { + peer := signedVerifiedVoterPeerVote(t, e, v.sets[7], key, uint16(rank+1), vote) + _, err := e.acceptVerifiedVoteResult(peer) + require.NoError(t, err) + } + } + require.Equal(t, id.Slot, e.ensureChain().Snapshot().LatestDirectFinalizedBlock.Slot) +} + +func TestAlpenglowVoterReplaysFourBlocksAfterNetworkFinality(t *testing.T) { + for _, reserved := range []bool{false, true} { + name := "synchronous" + if reserved { + name = "reserved" + } + t.Run(name, func(t *testing.T) { + e, v := newOrderingTestVoter(t, reserved) + for slot := uint64(40); slot <= 43; slot++ { + id := observeOrderingBlock(t, e, slot) + finalizeOrderingBlock(t, e, v, id) + drainOrderingEvents(t, v) + require.Equal(t, slot, v.highestFinal) + require.Less(t, v.admissionFloor(), slot) + require.False(t, v.history.VotedAt(slot), "network finality does not prove local execution") + before := v.snapshot() + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: slot, Source: "ordering-test"})) + drainOrderingEvents(t, v) + hash, ok := v.history.NotarizedVote(slot) + require.True(t, ok, "replay must still notarize after slow finality") + require.Equal(t, id.Hash, hash) + require.Greater(t, v.snapshot().BroadcastMessagesQueued, before.BroadcastMessagesQueued) + require.NoError(t, e.AlpenglowSafetyError()) + } + }) + } +} + +func TestAlpenglowVoterNetworkFinalityDuringLocalAdmission(t *testing.T) { + e, v := newOrderingTestVoter(t, false) + id := observeOrderingBlock(t, e, 40) + called := false + v.beforeLocalVoteInject = func(vote alpenglow.Vote) { + if called { + return + } + called = true + require.Equal(t, alpenglow.NewNotarizationVote(id.Slot, id.Hash), vote) + finalizeOrderingBlock(t, e, v, id) + } + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: id.Slot})) + drainOrderingEvents(t, v) + require.True(t, called) + require.Positive(t, v.snapshot().VotesCastThisRun) + message, _, err := v.sign(alpenglow.NewNotarizationVote(id.Slot, id.Hash), false) + require.NoError(t, err) + require.True(t, e.ensurePool().HasVerifiedVote(message)) + require.NoError(t, e.AlpenglowSafetyError()) +} + +func TestAlpenglowVoterDurableRootCannotOvertakeQueuedReplay(t *testing.T) { + e, v := newOrderingTestVoter(t, true) + id := observeOrderingBlock(t, e, 40) + finalizeOrderingBlock(t, e, v, id) + drainOrderingEvents(t, v) + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: id.Slot})) + e.PruneAlpenglowBefore(id.Slot) + require.Equal(t, uint64(39), e.ensurePool().Snapshot().RootSlot) + require.Contains(t, e.executedReplayBlocks, id, "queued replay must retain its execution proof") + queuedBefore := v.snapshot().BroadcastMessagesQueued + drainOrderingEvents(t, v) + require.Greater(t, v.snapshot().BroadcastMessagesQueued, queuedBefore) + require.Equal(t, id.Slot, v.history.Root) + require.Equal(t, id.Slot, e.ensurePool().Snapshot().RootSlot) + require.NotContains(t, e.executedReplayBlocks, id) + _, ok := v.history.NotarizedVote(id.Slot) + require.True(t, ok, "root vote must remain available to its intra-window child") + child := observeOrderingBlock(t, e, 41) + finalizeOrderingBlock(t, e, v, child) + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: child.Slot})) + drainOrderingEvents(t, v) + hash, ok := v.history.NotarizedVote(child.Slot) + require.True(t, ok) + require.Equal(t, child.Hash, hash) + require.NoError(t, e.AlpenglowSafetyError()) +} + +func TestAlpenglowVoterFinalityPreservesEarlierSkipDecision(t *testing.T) { + e, v := newOrderingTestVoter(t, true) + require.NoError(t, v.history.AddVote(alpenglow.NewSkipVote(40))) + id := observeOrderingBlock(t, e, 40) + finalizeOrderingBlock(t, e, v, id) + drainOrderingEvents(t, v) + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: 40})) + drainOrderingEvents(t, v) + require.True(t, v.history.HasSkipped(40)) + _, ok := v.history.NotarizedVote(40) + require.False(t, ok, "late replay must never replace an earlier round-one decision") + require.NoError(t, e.AlpenglowSafetyError()) +} + +func TestReservedRecoveryUsesVerifiedFinalityNotLiveAdmissionFloor(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 40) + h := v.reservation.through.Load() + crashReservedTestVoter(t, v) + cfg.InitializeVoteReservation = false + v, err = openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + _, _, err = v.sign(alpenglow.NewSkipVote(h+1), false) + require.ErrorIs(t, err, errVoterNotReady) + id := observeOrderingBlock(t, v.engine, h) + finalizeOrderingBlock(t, v.engine, v, id) + require.Less(t, v.admissionFloor(), h) + require.Equal(t, h, v.engine.alpenglowVerifiedFinalityFloor()) + reserveThrough(t, v.reservation, h+1) + for _, normal := range []bool{false, true} { + _, _, err = v.sign(alpenglow.NewSkipVote(h), normal) + require.ErrorIs(t, err, errVoterNotReady) + _, _, err = v.sign(alpenglow.NewSkipVote(h+1), normal) + require.NoError(t, err) + } +} diff --git a/pkg/consensus/voter_wait_slot_test.go b/pkg/consensus/voter_wait_slot_test.go new file mode 100644 index 000000000..b06a35819 --- /dev/null +++ b/pkg/consensus/voter_wait_slot_test.go @@ -0,0 +1,91 @@ +package consensus + +import ( + "crypto/ed25519" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func waitSlotTestVoter(t *testing.T, cutoff uint64) *alpenglowVoter { + t.Helper() + identity, authorized := voterTestKey(11), voterTestKey(12) + voteAccount := solana.PublicKey(voterTestKey(13).Public().(ed25519.PublicKey)) + set := voterTestValidatorSet(t, identity, authorized, voteAccount) + root := alpenglow.BlockID{Slot: 39, Hash: solana.Hash{0x39}} + engine, err := NewEngine(Config{AlpenglowShredVersion: 0x1234, AlpenglowIdentity: identity}) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, engine.Close()) }) + engine.SetAlpenglowEpochLookup(func(uint64) uint64 { return set.Epoch }) + require.NoError(t, engine.SetAlpenglowValidatorSet(set)) + engine.SetAlpenglowRoot(root) + voter, err := newAlpenglowVoterUnstarted(engine, VotingConfig{ + Identity: identity, AuthorizedVoter: authorized, VoteAccount: voteAccount, + HistoryDir: t.TempDir(), EpochForSlot: func(uint64) uint64 { return set.Epoch }, + Peers: func([]alpenglow.ValidatorStake) []alpenglow.VotorPeer { return nil }, + WaitToVoteSlot: cutoff, ReadyToVote: func(uint64) bool { return true }, + }, root, []alpenglow.ValidatorSet{set}) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, voter.close()) }) + return voter +} + +func TestWaitToVoteSlotGatesEveryNewVoteType(t *testing.T) { + voter := waitSlotTestVoter(t, 44) + constructors := []func(uint64) alpenglow.Vote{ + func(slot uint64) alpenglow.Vote { return alpenglow.NewNotarizationVote(slot, solana.Hash{1}) }, + alpenglow.NewFinalizationVote, + alpenglow.NewSkipVote, + func(slot uint64) alpenglow.Vote { return alpenglow.NewNotarizationFallbackVote(slot, solana.Hash{1}) }, + alpenglow.NewSkipFallbackVote, + } + for _, started := range []bool{false, true} { + voter.votingStarted = started + for _, voteAt := range constructors { + vote := voteAt(43) + _, _, err := voter.sign(vote, true) + require.ErrorIs(t, err, errVoterNotReady, "%s started=%t", vote.Type, started) + for _, slot := range []uint64{44, 45} { + message, _, err := voter.sign(voteAt(slot), true) + require.NoError(t, err, "%s slot=%d started=%t", vote.Type, slot, started) + require.Equal(t, voteAt(slot), message.Vote) + } + } + } +} + +func TestWaitToVoteSlotSplitsSkipWindowAndPersistsOnlyAllowedVotes(t *testing.T) { + voter := waitSlotTestVoter(t, 42) + require.NoError(t, voter.trySkipWindow(40)) + for _, slot := range []uint64{40, 41} { + require.False(t, voter.history.VotedAt(slot)) + } + for _, slot := range []uint64{42, 43} { + require.True(t, voter.history.HasSkipped(slot)) + } + restored, err := alpenglow.LoadVoteHistory(voter.historyDir, voter.node) + require.NoError(t, err) + require.Equal(t, voter.history.VotesCast, restored.VotesCast) + require.EqualValues(t, 2, voter.engine.ensurePool().Snapshot().VerifiedVotes) + // Joining live voting must not make older slots eligible afterward. + require.True(t, voter.votingStarted) + voted, err := voter.cast(alpenglow.NewSkipVote(41), false) + require.NoError(t, err) + require.False(t, voted) + require.False(t, voter.history.VotedAt(41)) +} + +func TestWaitToVoteSlotPreservesAuthenticatedHistoryRestoration(t *testing.T) { + voter := waitSlotTestVoter(t, 44) + require.NoError(t, voter.history.AddVote(alpenglow.NewSkipVote(40))) + require.NoError(t, voter.saveHistory()) + restored, err := alpenglow.LoadVoteHistory(voter.historyDir, voter.node) + require.NoError(t, err) + voter.history = restored + require.NoError(t, voter.restoreVotesForEpoch(voter.epochForSlot(40))) + require.EqualValues(t, 1, voter.engine.ensurePool().Snapshot().VerifiedVotes) + require.False(t, voter.votingStarted, "restoring a recorded vote must not bypass startup readiness") + require.ErrorIs(t, voter.votingGateError(41), errVoterNotReady) +} diff --git a/pkg/costmodel/costmodel_test.go b/pkg/costmodel/costmodel_test.go index f5e41ec45..82160999c 100644 --- a/pkg/costmodel/costmodel_test.go +++ b/pkg/costmodel/costmodel_test.go @@ -81,15 +81,6 @@ func TestEstimateTransactionCostCountsPrecompileSignatures(t *testing.T) { ) } -func TestLimitsForFeaturesRaiseBlockLimitsTo100m(t *testing.T) { - feats := features.NewFeaturesDefault() - assert.Equal(t, uint64(MaxBlockUnitsSIMD0256), LimitsForFeatures(feats).BlockCost) - - feats.EnableFeature(features.RaiseBlockLimitsTo100m, 123) - assert.Equal(t, uint64(MaxBlockUnitsSIMD0286), LimitsForFeatures(feats).BlockCost) - assert.Equal(t, uint64(MaxBlockUnitsSIMD0256), DefaultLimits().BlockCost) -} - func TestWritableAccountsUsesUnsignedWritableRange(t *testing.T) { tx := &solana.Transaction{Message: solana.Message{ Header: solana.MessageHeader{ @@ -284,3 +275,12 @@ func TestCostTrackerAcceptsUnderLimits(t *testing.T) { assert.Equal(t, ExceedNone, tracker.WouldExceed(cost)) _ = wire } + +func TestPackEntryBytesMaxChargesEveryFECSetInBatch(t *testing.T) { + // One initial set, two reserved ending sets, then exactly one full batch. + shreds := uint64((1 + 2 + FECSetsPerBatch) * DataShredsPerFECSet) + want := uint64(DefaultTargetBatchBytes - 8 - MaxMicroblockBytes) + assert.Equal(t, want, PackEntryBytesMax(shreds, MaxMicroblockBytes)) + assert.Equal(t, want, PackEntryBytesMax(shreds+DataShredsPerFECSet-1, MaxMicroblockBytes)) + assert.Zero(t, PackEntryBytesMax(shreds-DataShredsPerFECSet, MaxMicroblockBytes)) +} diff --git a/pkg/costmodel/entry_bytes.go b/pkg/costmodel/entry_bytes.go index ab91b8d5a..ebf89fc47 100644 --- a/pkg/costmodel/entry_bytes.go +++ b/pkg/costmodel/entry_bytes.go @@ -1,7 +1,7 @@ package costmodel // PackEntryBytesMax is the shred-safe entry-byte bound: we close a batch when -// the next microblock would not fit in one FEC set, so padding is at most one +// the next microblock would not fit in FECSetsPerBatch FEC sets, so padding is at most one // microblock. The slot cap is still min(this, SIMD-0525), decided at schedule // time like Firedancer pack. // @@ -23,13 +23,16 @@ func PackEntryBytesMax(slotMaxDataShreds, maxMicroblock uint64) uint64 { } middle := fecSets - first - lastFEC minBatch := wmark - maxMicroblock - return middle * minBatch + return (middle / FECSetsPerBatch) * minBatch } // DefaultPackEntryBytes is min(shred-safe, SIMD-0525) minus one ending tick. func DefaultPackEntryBytes() uint64 { - shredSafe := PackEntryBytesMax(DefaultMaxDataShredsPerSlot, MaxMicroblockBytes) - cap := uint64(DefaultMaxEntryBytesPerSlot) + return packEntryBytes(DefaultMaxDataShredsPerSlot, DefaultMaxEntryBytesPerSlot) +} + +func packEntryBytes(maxDataShreds, cap uint64) uint64 { + shredSafe := PackEntryBytesMax(maxDataShreds, MaxMicroblockBytes) if shredSafe > 0 && shredSafe < cap { cap = shredSafe } diff --git a/pkg/costmodel/limits.go b/pkg/costmodel/limits.go index 2e4766569..0afdf925f 100644 --- a/pkg/costmodel/limits.go +++ b/pkg/costmodel/limits.go @@ -1,7 +1,11 @@ package costmodel import ( + "fmt" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/safemath" + "github.com/Overclock-Validator/mithril/pkg/sealevel" "github.com/gagliardetto/solana-go" ) @@ -47,11 +51,11 @@ const ( TypicalDataShredPayloadBytes = 963 // DataShredsPerFECSet matches turbine's 32:32 erasure batch. DataShredsPerFECSet = 32 - // FECSetsPerBatch is the close watermark: hold until one FEC set is full. - FECSetsPerBatch = 1 + // FECSetsPerBatch is the close watermark: hold until two FEC sets are full. + FECSetsPerBatch = 2 // TypicalFECSetPayloadBytes is one full unsigned FEC set. TypicalFECSetPayloadBytes = DataShredsPerFECSet * TypicalDataShredPayloadBytes - // DefaultTargetBatchBytes is one FEC set. A short leftover is only + // DefaultTargetBatchBytes is two FEC sets. A short leftover is only // emitted at slot end (Freeze / ending tick). DefaultTargetBatchBytes = FECSetsPerBatch * TypicalFECSetPayloadBytes ) @@ -75,11 +79,53 @@ func DefaultLimits() Limits { } } -// LimitsForFeatures returns the cost limits selected by the bank's feature set. +// LimitsForSlot to apply slot-time reductions at the correct epoch boundary. func LimitsForFeatures(feats *features.Features) Limits { limits := DefaultLimits() if feats != nil && feats.IsActive(features.RaiseBlockLimitsTo100m) { limits.BlockCost = MaxBlockUnitsSIMD0286 + limits.WritableAccountCost = 40_000_000 } return limits } + +// LimitsForSlot mirrors Agave v4.3.0-rc.1 runtime/src/slot_params.rs. +// Reference: https://github.com/anza-xyz/agave/blob/v4.3.0-rc.1/runtime/src/slot_params.rs A slot-time +// gate takes effect in the epoch after activation; among effective gates the +// shortest duration wins, even if longer-duration gates activate later. +func LimitsForSlot(feats *features.Features, schedule *sealevel.SysvarEpochSchedule, slot uint64) (Limits, error) { + limits := DefaultLimits() + for _, transition := range []struct { + gate features.FeatureGate + account, block, data, shreds, entries uint64 + }{ + {features.ReduceSlotTimeTo350ms, 21_000_000, 52_500_000, 87_500_000, 28_672, 18_350_080}, + {features.ReduceSlotTimeTo300ms, 18_000_000, 45_000_000, 75_000_000, 24_576, 15_728_640}, + {features.ReduceSlotTimeTo250ms, 15_000_000, 37_500_000, 62_500_000, 20_480, 13_107_200}, + {features.ReduceSlotTimeTo200ms, 12_000_000, 30_000_000, 50_000_000, 16_384, 10_485_760}, + } { + if feats == nil { + break + } + activation, active := feats.ActivationSlot(transition.gate) + if !active { + continue + } + if schedule == nil || schedule.SlotsPerEpoch == 0 { + return Limits{}, fmt.Errorf("epoch schedule required for slot-time cost limits") + } + effective := schedule.FirstSlotInEpoch(safemath.SaturatingAddU64(schedule.GetEpoch(activation), 1)) + if effective > slot { + continue + } + limits.WritableAccountCost = transition.account + limits.BlockCost = transition.block + limits.AllocatedDataSizeDelta = transition.data + limits.MaxEntryBytes = packEntryBytes(transition.shreds, transition.entries) + } + if feats != nil && feats.IsActive(features.RaiseBlockLimitsTo100m) { + limits.BlockCost = limits.BlockCost * 100 / 60 + limits.WritableAccountCost = limits.WritableAccountCost * 100 / 60 + } + return limits, nil +} diff --git a/pkg/costmodel/limits_test.go b/pkg/costmodel/limits_test.go new file mode 100644 index 000000000..60de4062a --- /dev/null +++ b/pkg/costmodel/limits_test.go @@ -0,0 +1,13 @@ +package costmodel + +import "testing" + +func TestDefaultTargetBatchBytesMatchesTwoTypicalFECSets(t *testing.T) { + const want = 61_632 + if DefaultTargetBatchBytes != want { + t.Fatalf("DefaultTargetBatchBytes = %d, want %d", DefaultTargetBatchBytes, want) + } + if DefaultTargetBatchBytes != 2*DataShredsPerFECSet*TypicalDataShredPayloadBytes { + t.Fatal("default batch target must remain two complete typical FEC payloads") + } +} diff --git a/pkg/costmodel/slot_limits_test.go b/pkg/costmodel/slot_limits_test.go new file mode 100644 index 000000000..0d521e744 --- /dev/null +++ b/pkg/costmodel/slot_limits_test.go @@ -0,0 +1,88 @@ +package costmodel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/stretchr/testify/require" +) + +func TestSlotLimitsMatchAgaveTable(t *testing.T) { + schedule := &sealevel.SysvarEpochSchedule{SlotsPerEpoch: 100} + for _, tc := range []struct { + name string + gate features.FeatureGate + account, block, data, shreds, entries uint64 + }{ + {"400ms", features.FeatureGate{}, 24_000_000, 60_000_000, 100_000_000, 32768, 20 * 1024 * 1024}, + {"350ms", features.ReduceSlotTimeTo350ms, 21_000_000, 52_500_000, 87_500_000, 28672, 18_350_080}, + {"300ms", features.ReduceSlotTimeTo300ms, 18_000_000, 45_000_000, 75_000_000, 24576, 15_728_640}, + {"250ms", features.ReduceSlotTimeTo250ms, 15_000_000, 37_500_000, 62_500_000, 20480, 13_107_200}, + {"200ms", features.ReduceSlotTimeTo200ms, 12_000_000, 30_000_000, 50_000_000, 16384, 10_485_760}, + } { + t.Run(tc.name, func(t *testing.T) { + f := features.NewFeaturesDefault() + if tc.name != "400ms" { + f.EnableFeature(tc.gate, 50) + } + before, err := LimitsForSlot(f, schedule, 99) + require.NoError(t, err) + require.Equal(t, DefaultLimits(), before) + for _, raise := range []bool{false, true} { + if raise { + f.EnableFeature(features.RaiseBlockLimitsTo100m, 0) + } + got, err := LimitsForSlot(f, schedule, 100) + require.NoError(t, err) + account, block := tc.account, tc.block + if raise { + account = account * 100 / 60 + block = block * 100 / 60 + } + require.Equal(t, account, got.WritableAccountCost) + require.Equal(t, block, got.BlockCost) + require.Equal(t, tc.data, got.AllocatedDataSizeDelta) + require.Equal(t, tc.entries-EntryHeaderBytes, got.MaxEntryBytes) + require.LessOrEqual(t, got.MaxEntryBytes, PackEntryBytesMax(tc.shreds, MaxMicroblockBytes)) + require.Equal(t, uint64(DefaultTargetBatchBytes), got.MaxBatchBytes) + } + }) + } +} + +func TestSlotLimitsDoNotLengthenSlotsForLaterGates(t *testing.T) { + f := features.NewFeaturesDefault() + f.EnableFeature(features.ReduceSlotTimeTo200ms, 50) + f.EnableFeature(features.ReduceSlotTimeTo350ms, 150) + schedule := &sealevel.SysvarEpochSchedule{SlotsPerEpoch: 100} + for _, slot := range []uint64{100, 199, 200, 400} { + limits, err := LimitsForSlot(f, schedule, slot) + require.NoError(t, err) + require.Equal(t, uint64(30_000_000), limits.BlockCost) + } +} + +func TestSlotLimitsActivationDuringWarmup(t *testing.T) { + f := features.NewFeaturesDefault() + f.EnableFeature(features.ReduceSlotTimeTo200ms, 40) + // Epoch 0: [0,32), epoch 1: [32,96), normal epoch 2: [96,224). + schedule := &sealevel.SysvarEpochSchedule{Warmup: true, SlotsPerEpoch: 128, FirstNormalEpoch: 2, FirstNormalSlot: 96} + before, err := LimitsForSlot(f, schedule, 95) + require.NoError(t, err) + require.Equal(t, uint64(60_000_000), before.BlockCost) + after, err := LimitsForSlot(f, schedule, 96) + require.NoError(t, err) + require.Equal(t, uint64(30_000_000), after.BlockCost) +} + +func TestSlotLimitsRequireScheduleForActiveReductions(t *testing.T) { + f := features.NewFeaturesDefault() + _, err := LimitsForSlot(f, nil, 10) + require.NoError(t, err) + f.EnableFeature(features.ReduceSlotTimeTo200ms, 0) + _, err = LimitsForSlot(f, nil, 10) + require.Error(t, err) + _, err = LimitsForSlot(f, &sealevel.SysvarEpochSchedule{}, 10) + require.Error(t, err) +} diff --git a/pkg/costmodel/transaction_cost.go b/pkg/costmodel/transaction_cost.go index 1dd4f0c71..3a804b037 100644 --- a/pkg/costmodel/transaction_cost.go +++ b/pkg/costmodel/transaction_cost.go @@ -55,6 +55,13 @@ func EstimateTransactionCost(tx *solana.Transaction, feats *features.Features) ( }, nil } + return EstimatePreparedTransactionCost(tx, instrs, limits, feats), nil +} + +// EstimatePreparedTransactionCost reuses successfully parsed instructions and +// compute limits from the same immutable transaction and feature snapshot. +func EstimatePreparedTransactionCost(tx *solana.Transaction, instrs []sealevel.Instruction, limits *sealevel.ComputeBudgetLimits, feats *features.Features) TransactionCost { + writable := writableAccounts(tx) loadedDataCost := loadedAccountsDataSizeCost(limits.LoadedAccountBytes) // Banking-stage admission must reserve at least one page for the fee // payer, including V1 transactions whose inline loaded-data limit is zero. @@ -62,13 +69,13 @@ func EstimateTransactionCost(tx *solana.Transaction, feats *features.Features) ( loadedDataCost = max(loadedDataCost, uint64(HeapCost)) return TransactionCost{ SignatureCost: signatureCost(tx, instrs, feats), - WriteLockCost: writeLockCost(countWriteLocks(tx)), + WriteLockCost: writeLockCost(uint64(len(writable))), DataBytesCost: instructionDataCost(tx), ProgramsExecutionCost: uint64(limits.ComputeUnitLimit), LoadedAccountsDataSizeCost: loadedDataCost, AllocatedAccountsDataSize: estimateAllocDelta(instrs, feats), - WritableAccounts: writableAccounts(tx), - }, nil + WritableAccounts: writable, + } } func signatureCost(tx *solana.Transaction, instrs []sealevel.Instruction, feats *features.Features) uint64 { @@ -177,7 +184,7 @@ func replayInstrsAndAcctMetas(tx *solana.Transaction, feats *features.Features) } upgradeableLoaderPresent := false for _, key := range tx.Message.AccountKeys { - if key.String() == "BPFLoaderUpgradeab1e11111111111111111111111" { + if key == addresses.BpfLoaderUpgradeableAddr { upgradeableLoaderPresent = true break } diff --git a/pkg/cu/cu.go b/pkg/cu/cu.go index 79d337199..f90aa1e2c 100644 --- a/pkg/cu/cu.go +++ b/pkg/cu/cu.go @@ -33,6 +33,11 @@ func (cm *ComputeMeter) Consume(cost uint64) error { return nil } +// Disabled reports whether metering is currently switched off (Consume is a no-op). +func (cm *ComputeMeter) Disabled() bool { + return cm.disable +} + func (cm *ComputeMeter) Used() uint64 { return cm.startingBalance - cm.computeMeter } diff --git a/pkg/global/global_ctx.go b/pkg/global/global_ctx.go index f8241f771..d99d9c5c6 100644 --- a/pkg/global/global_ctx.go +++ b/pkg/global/global_ctx.go @@ -104,9 +104,19 @@ func PendingStakeEntriesSnapshot() []accountsdb.StakeIndexEntry { return out } +// DropPendingStakePubkeys drops only the discarded bank's slot. A local leader +// can have pending entries at later slots even while replay is behind it. +func DropPendingStakePubkeys(slot uint64) int { + instance.pendingStakeMutex.Lock() + defer instance.pendingStakeMutex.Unlock() + dropped := len(instance.pendingStakeBySlot[slot]) + delete(instance.pendingStakeBySlot, slot) + return dropped +} + // DropPendingStakePubkeysFrom discards pending entries for slots >= fromSlot. -// Called by the fork-switch unwind so wrong-fork stake entries never reach the -// durable index. Returns the number of entries dropped. +// This is a global reset operation. Per-bank discard and replay unwind must +// use DropPendingStakePubkeys so they preserve independent leader banks. func DropPendingStakePubkeysFrom(fromSlot uint64) int { instance.pendingStakeMutex.Lock() defer instance.pendingStakeMutex.Unlock() diff --git a/pkg/lthash/lthash.go b/pkg/lthash/lthash.go index 9c4aaee1b..faa09a8a5 100644 --- a/pkg/lthash/lthash.go +++ b/pkg/lthash/lthash.go @@ -120,16 +120,15 @@ func (ltHash *LtHash) Clone() *LtHash { return new } +// MixIn adds other's 1024 lanes to ltHash, lane-wise modulo 2^16. The +// lane arithmetic is vectorized where the platform supports it (see mix.go). func (ltHash *LtHash) MixIn(other *LtHash) { - for i := range numElements { - ltHash.value[i] = ltHash.value[i] + other.value[i] - } + mixIn(<Hash.value, &other.value) } +// MixOut subtracts other's lanes from ltHash, the inverse of MixIn. func (ltHash *LtHash) MixOut(other *LtHash) { - for i := range numElements { - ltHash.value[i] = ltHash.value[i] - other.value[i] - } + mixOut(<Hash.value, &other.value) } func (ltHash *LtHash) Add(other *LtHash) *LtHash { @@ -143,12 +142,7 @@ func (ltHash *LtHash) Sub(other *LtHash) *LtHash { } func (ltHash *LtHash) Equals(other *LtHash) bool { - for i, element := range ltHash.value { - if element != other.value[i] { - return false - } - } - return true + return ltHash.value == other.value } func (ltHash *LtHash) Checksum() []byte { diff --git a/pkg/lthash/mix.go b/pkg/lthash/mix.go new file mode 100644 index 000000000..6cce9a4d9 --- /dev/null +++ b/pkg/lthash/mix.go @@ -0,0 +1,16 @@ +package lthash + +// mixInGeneric and mixOutGeneric are the portable lane loops. Every +// architecture-specific implementation must produce identical results: the +// lanes are independent uint16 additions and subtractions modulo 2^16. +func mixInGeneric(dst, src *[numElements]uint16) { + for i := range numElements { + dst[i] += src[i] + } +} + +func mixOutGeneric(dst, src *[numElements]uint16) { + for i := range numElements { + dst[i] -= src[i] + } +} diff --git a/pkg/lthash/mix_amd64.go b/pkg/lthash/mix_amd64.go new file mode 100644 index 000000000..f4818245c --- /dev/null +++ b/pkg/lthash/mix_amd64.go @@ -0,0 +1,32 @@ +//go:build amd64 && !purego + +package lthash + +import "golang.org/x/sys/cpu" + +// useAVX2 selects the vector lane loops. cpu.X86.HasAVX2 already includes +// the operating-system XSAVE/YMM-state check. Tests flip it to compare the +// two implementations on the same machine. +var useAVX2 = cpu.X86.HasAVX2 + +func mixIn(dst, src *[numElements]uint16) { + if useAVX2 { + mixInAVX2(dst, src) + return + } + mixInGeneric(dst, src) +} + +func mixOut(dst, src *[numElements]uint16) { + if useAVX2 { + mixOutAVX2(dst, src) + return + } + mixOutGeneric(dst, src) +} + +//go:noescape +func mixInAVX2(dst, src *[numElements]uint16) + +//go:noescape +func mixOutAVX2(dst, src *[numElements]uint16) diff --git a/pkg/lthash/mix_amd64.s b/pkg/lthash/mix_amd64.s new file mode 100644 index 000000000..2233de57e --- /dev/null +++ b/pkg/lthash/mix_amd64.s @@ -0,0 +1,56 @@ +//go:build amd64 && !purego + +#include "textflag.h" + +// The LtHash value is 1024 uint16 lanes = 2048 bytes = 16 iterations of +// four 32-byte YMM vectors. VPADDW/VPSUBW operate on 16-bit lanes modulo +// 2^16, exactly like the generic Go loop. Loads and stores are unaligned +// (VMOVDQU): LtHash values live inside Go structs with 2-byte alignment. + +// func mixInAVX2(dst, src *[1024]uint16) +TEXT ·mixInAVX2(SB), NOSPLIT, $0-16 + MOVQ dst+0(FP), DI + MOVQ src+8(FP), SI + XORQ AX, AX +mixin_loop: + VMOVDQU (DI)(AX*1), Y0 + VMOVDQU 32(DI)(AX*1), Y1 + VMOVDQU 64(DI)(AX*1), Y2 + VMOVDQU 96(DI)(AX*1), Y3 + VPADDW (SI)(AX*1), Y0, Y0 + VPADDW 32(SI)(AX*1), Y1, Y1 + VPADDW 64(SI)(AX*1), Y2, Y2 + VPADDW 96(SI)(AX*1), Y3, Y3 + VMOVDQU Y0, (DI)(AX*1) + VMOVDQU Y1, 32(DI)(AX*1) + VMOVDQU Y2, 64(DI)(AX*1) + VMOVDQU Y3, 96(DI)(AX*1) + ADDQ $128, AX + CMPQ AX, $2048 + JB mixin_loop + VZEROUPPER + RET + +// func mixOutAVX2(dst, src *[1024]uint16) +TEXT ·mixOutAVX2(SB), NOSPLIT, $0-16 + MOVQ dst+0(FP), DI + MOVQ src+8(FP), SI + XORQ AX, AX +mixout_loop: + VMOVDQU (DI)(AX*1), Y0 + VMOVDQU 32(DI)(AX*1), Y1 + VMOVDQU 64(DI)(AX*1), Y2 + VMOVDQU 96(DI)(AX*1), Y3 + VPSUBW (SI)(AX*1), Y0, Y0 + VPSUBW 32(SI)(AX*1), Y1, Y1 + VPSUBW 64(SI)(AX*1), Y2, Y2 + VPSUBW 96(SI)(AX*1), Y3, Y3 + VMOVDQU Y0, (DI)(AX*1) + VMOVDQU Y1, 32(DI)(AX*1) + VMOVDQU Y2, 64(DI)(AX*1) + VMOVDQU Y3, 96(DI)(AX*1) + ADDQ $128, AX + CMPQ AX, $2048 + JB mixout_loop + VZEROUPPER + RET diff --git a/pkg/lthash/mix_amd64_test.go b/pkg/lthash/mix_amd64_test.go new file mode 100644 index 000000000..110c1b39a --- /dev/null +++ b/pkg/lthash/mix_amd64_test.go @@ -0,0 +1,74 @@ +//go:build amd64 && !purego + +package lthash + +import ( + "math/rand" + "testing" + "unsafe" +) + +// TestMixAVX2AgainstGeneric runs the assembly directly (when the CPU has +// AVX2) against the portable loops so the comparison does not depend on the +// dispatch variable. +func TestMixAVX2AgainstGeneric(t *testing.T) { + if !useAVX2 { + t.Skip("no AVX2 on this machine") + } + rng := rand.New(rand.NewSource(6)) + for iter := 0; iter < 2000; iter++ { + dst := randomLanes(rng) + src := randomLanes(rng) + want, got := *dst, *dst + mixInGeneric(&want, src) + mixInAVX2(&got, src) + if got != want { + t.Fatalf("mixInAVX2 diverges (iteration %d)", iter) + } + want, got = *dst, *dst + mixOutGeneric(&want, src) + mixOutAVX2(&got, src) + if got != want { + t.Fatalf("mixOutAVX2 diverges (iteration %d)", iter) + } + } +} + +// TestMixGenericFallbackSelectable makes sure the dispatch honours the flag, +// so a machine without AVX2 takes the portable path. +func TestMixGenericFallbackSelectable(t *testing.T) { + saved := useAVX2 + defer func() { useAVX2 = saved }() + useAVX2 = false + rng := rand.New(rand.NewSource(8)) + dst := randomLanes(rng) + src := randomLanes(rng) + want := *dst + mixInGeneric(&want, src) + mixIn(dst, src) + if *dst != want { + t.Fatal("generic fallback must be used when AVX2 is disabled") + } +} + +func TestMixAVX2UnalignedAndAliased(t *testing.T) { + if !useAVX2 { + t.Skip("AVX2 unavailable") + } + rng := rand.New(rand.NewSource(19)) + for off := 0; off < 32; off += 2 { + storage := make([]byte, numElements*2+32) + dst := (*[numElements]uint16)(unsafe.Pointer(&storage[off])) + *dst = *randomLanes(rng) + want := *dst + mixInGeneric(&want, &want) + mixInAVX2(dst, dst) + if *dst != want { + t.Fatalf("aliased addition offset %d", off) + } + mixOutAVX2(dst, dst) + if *dst != ([numElements]uint16{}) { + t.Fatalf("aliased subtraction offset %d", off) + } + } +} diff --git a/pkg/lthash/mix_generic.go b/pkg/lthash/mix_generic.go new file mode 100644 index 000000000..79e7b0bb4 --- /dev/null +++ b/pkg/lthash/mix_generic.go @@ -0,0 +1,6 @@ +//go:build !amd64 || purego + +package lthash + +func mixIn(dst, src *[numElements]uint16) { mixInGeneric(dst, src) } +func mixOut(dst, src *[numElements]uint16) { mixOutGeneric(dst, src) } diff --git a/pkg/lthash/mix_test.go b/pkg/lthash/mix_test.go new file mode 100644 index 000000000..272051689 --- /dev/null +++ b/pkg/lthash/mix_test.go @@ -0,0 +1,116 @@ +package lthash + +import ( + "math/rand" + "testing" +) + +func randomLanes(rng *rand.Rand) *[numElements]uint16 { + var lanes [numElements]uint16 + for i := range lanes { + switch rng.Intn(8) { + case 0: + lanes[i] = 0 + case 1: + lanes[i] = 0xffff + case 2: + lanes[i] = 0x8000 + default: + lanes[i] = uint16(rng.Uint32()) + } + } + return &lanes +} + +// TestMixMatchesGeneric checks the platform mixIn/mixOut against the +// portable loops, including wrap-around lanes, and that MixOut inverts MixIn. +func TestMixMatchesGeneric(t *testing.T) { + rng := rand.New(rand.NewSource(3)) + for iter := 0; iter < 2000; iter++ { + dst := randomLanes(rng) + src := randomLanes(rng) + wantIn := *dst + mixInGeneric(&wantIn, src) + gotIn := *dst + mixIn(&gotIn, src) + if gotIn != wantIn { + t.Fatalf("mixIn diverges from the generic loop (iteration %d)", iter) + } + wantOut := *dst + mixOutGeneric(&wantOut, src) + gotOut := *dst + mixOut(&gotOut, src) + if gotOut != wantOut { + t.Fatalf("mixOut diverges from the generic loop (iteration %d)", iter) + } + roundTrip := gotIn + mixOut(&roundTrip, src) + if roundTrip != *dst { + t.Fatalf("mixOut does not invert mixIn (iteration %d)", iter) + } + } + // In-place: mixing a value into itself doubles every lane. + dst := randomLanes(rng) + want := *dst + for i := range want { + want[i] *= 2 + } + mixIn(dst, dst) + if *dst != want { + t.Fatal("mixIn with aliased operands must double every lane") + } + mixOut(dst, dst) + if *dst != [numElements]uint16{} { + t.Fatal("mixOut with aliased operands must clear every lane") + } +} + +func TestLtHashMixInMixOutAndEquals(t *testing.T) { + rng := rand.New(rand.NewSource(4)) + var a, b, c LtHash + a.value = *randomLanes(rng) + b.value = *randomLanes(rng) + c = *a.Clone() + c.MixIn(&b) + if c.Equals(&a) { + t.Fatal("mixing in a random value must change the hash") + } + c.MixOut(&b) + if !c.Equals(&a) { + t.Fatal("MixOut must undo MixIn") + } + c.value[numElements-1]++ + if c.Equals(&a) { + t.Fatal("Equals must see a last-lane difference") + } +} + +func BenchmarkMixIn(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixIn(dst, src) + } +} + +func BenchmarkMixInGeneric(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixInGeneric(dst, src) + } +} + +func BenchmarkMixOut(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixOut(dst, src) + } +} diff --git a/pkg/merkletree/merkletree.go b/pkg/merkletree/merkletree.go index 988f2633c..0de1d98f3 100644 --- a/pkg/merkletree/merkletree.go +++ b/pkg/merkletree/merkletree.go @@ -46,8 +46,29 @@ func (n *Nodes) GetRoot() (out *[32]byte) { return &n.Nodes[len(n.Nodes)-1] } -// TODO provide a method for memory-efficient Merkle construction when only the root is requested. -// Can be implemented using recursion root level downwards +// HashRoot computes the same root as HashNodes without retaining proof nodes. +// Empty input returns zero. Each level overwrites the preceding level, so only +// one hash per leaf is allocated. Leaves are never modified. +func HashRoot(leaves [][]byte) (root [32]byte) { + if len(leaves) == 0 { + return root + } + if len(leaves) == 1 { + return HashLeaf(leaves[0]) + } + nodes := make([][32]byte, len(leaves)) + for i, leaf := range leaves { + nodes[i] = HashLeaf(leaf) + } + for len(nodes) > 1 { + for i := 0; i < len(nodes); i += 2 { + right := min(i+1, len(nodes)-1) + nodes[i/2] = HashIntermediate(&nodes[i], &nodes[right]) + } + nodes = nodes[:(len(nodes)+1)/2] + } + return nodes[0] +} // HashNodes constructs proof data from a set of leaves. // @@ -97,6 +118,14 @@ func HashNodes(leaves [][]byte) (out Nodes) { // HashLeaf returns the hash of a leaf node. func HashLeaf(data []byte) (out [32]byte) { + if len(data) == 64 { + // Transaction signatures are fixed-width leaves. A single buffer avoids + // incremental hash writes while retaining the leaf domain separator. + var input [65]byte + input[0] = TypeLeaf + copy(input[1:], data) + return sha256.Sum256(input[:]) + } h := sha256.New() h.Write([]byte{TypeLeaf}) h.Write(data) @@ -106,12 +135,11 @@ func HashLeaf(data []byte) (out [32]byte) { // HashIntermediate returns the hash of an intermediate node. func HashIntermediate(left *[32]byte, right *[32]byte) (out [32]byte) { - h := sha256.New() - h.Write([]byte{TypeIntermediate}) - h.Write(left[:]) - h.Write(right[:]) - h.Sum(out[:0]) - return + var input [65]byte + input[0] = TypeIntermediate + copy(input[1:33], left[:]) + copy(input[33:], right[:]) + return sha256.Sum256(input[:]) } // nextLevelLen returns the amount of nodes in the layer above the current one, diff --git a/pkg/merkletree/root_test.go b/pkg/merkletree/root_test.go new file mode 100644 index 000000000..1437c3833 --- /dev/null +++ b/pkg/merkletree/root_test.go @@ -0,0 +1,60 @@ +package merkletree + +import ( + "bytes" + "crypto/sha256" + "fmt" + "testing" +) + +// Independent construction: retain separate levels and use one-shot SHA256 +// over explicit domain-prefixed bytes, without the production hash helpers. +func referenceRoot(leaves [][]byte) [32]byte { + if len(leaves) == 0 { + return [32]byte{} + } + level := make([][32]byte, len(leaves)) + for i, leaf := range leaves { + level[i] = sha256.Sum256(append([]byte{0}, leaf...)) + } + for len(level) > 1 { + next := make([][32]byte, (len(level)+1)/2) + for i := range next { + left, right := level[2*i], level[min(2*i+1, len(level)-1)] + data := append([]byte{1}, left[:]...) + data = append(data, right[:]...) + next[i] = sha256.Sum256(data) + } + level = next + } + return level[0] +} + +func TestHashRootMatchesCanonicalTree(t *testing.T) { + counts := []int{0, 1, 2, 3, 7, 8, 9, 31, 32, 33, 63, 64, 65, 127, 128, 129, 255, 256, 257, 311, 511, 512, 513, 1023, 1024, 1025} + for _, size := range []int{0, 31, 32, 55, 56, 63, 64, 65, 1232} { + for _, count := range counts { + t.Run(fmt.Sprintf("%dx%d", count, size), func(t *testing.T) { + leaves := make([][]byte, count) + for i := range leaves { + leaves[i] = make([]byte, size) + for j := range leaves[i] { + leaves[i][j] = byte(i*71 + i/256 + j*17) + } + } + before := bytes.Join(leaves, nil) + want := referenceRoot(leaves) + if got := HashRoot(leaves); got != want { + t.Fatalf("root %x, want %x", got, want) + } + nodes := HashNodes(leaves) + if got := nodes.GetRoot(); got != nil && *got != want { + t.Fatalf("proof root %x, want %x", *got, want) + } + if !bytes.Equal(before, bytes.Join(leaves, nil)) { + t.Fatal("hashing modified input leaves") + } + }) + } + } +} diff --git a/pkg/metrics/account_loader_test.go b/pkg/metrics/account_loader_test.go new file mode 100644 index 000000000..fed0c86ed --- /dev/null +++ b/pkg/metrics/account_loader_test.go @@ -0,0 +1,40 @@ +package metrics + +import ( + "reflect" + "testing" +) + +// Exercise every field so adding a metric without merging it is caught. +func TestAccountLoaderAccumulateAllFields(t *testing.T) { + var src AccountLoader + var fill func(reflect.Value) + fill = func(v reflect.Value) { + for i := 0; i < v.NumField(); i++ { + f := v.Field(i) + if f.Kind() == reflect.Struct { + fill(f) + } else { + f.SetUint(uint64(i + 1)) + } + } + } + fill(reflect.ValueOf(&src).Elem()) + var dst AccountLoader + dst.Accumulate(src) + if dst != src { + t.Fatal("first merge lost fields") + } + dst.Accumulate(src) + var check func(reflect.Value, reflect.Value) + check = func(a, b reflect.Value) { + for i := 0; i < a.NumField(); i++ { + if a.Field(i).Kind() == reflect.Struct { + check(a.Field(i), b.Field(i)) + } else if a.Field(i).Uint() != 2*b.Field(i).Uint() { + t.Errorf("field %s not accumulated", a.Type().Field(i).Name) + } + } + } + check(reflect.ValueOf(dst), reflect.ValueOf(src)) +} diff --git a/pkg/metrics/metrics.go b/pkg/metrics/metrics.go index 10548fdd4..fd807da0b 100644 --- a/pkg/metrics/metrics.go +++ b/pkg/metrics/metrics.go @@ -15,13 +15,25 @@ func (t *Timing) AddTiming(d time.Duration) { atomic.AddUint64(&t.SumNanoseconds, uint64(d.Nanoseconds())) } +// StartTiming avoids reading the clock when a caller does not record timings. +func StartTiming(enabled bool) time.Time { + if enabled { + return time.Now() + } + return time.Time{} +} + func (t *Timing) AddTimingSince(start time.Time) { - t.AddTiming(time.Since(start)) + if !start.IsZero() { + t.AddTiming(time.Since(start)) + } } // AccountLoader is the per-slot decomposition of LoadBlockAccounts. Counters // describe logical loader work; allocation counters cover objects/data created // directly by the batch loader rather than runtime or Pebble internals. +// Counts sum loader operations across groups, not unique keys/files per block; +// in particular UniqueAppendVecs sums each batch's distinct file count. type AccountLoader struct { AddressTableLookups Timing DedupeBlockAccounts Timing @@ -94,15 +106,123 @@ type AccountLoader struct { SysvarCachePublicationEpochRejects uint64 } -// TurbineIngress is the exact per-slot pre-replay pipeline decomposition. +// Accumulate merges completed loader work. Both records must be exclusively +// owned by the replay goroutine; this is not a concurrent snapshot operation. +func (dst *AccountLoader) Accumulate(src AccountLoader) { + dst.AddressTableLookups.Count += src.AddressTableLookups.Count + dst.AddressTableLookups.SumNanoseconds += src.AddressTableLookups.SumNanoseconds + dst.DedupeBlockAccounts.Count += src.DedupeBlockAccounts.Count + dst.DedupeBlockAccounts.SumNanoseconds += src.DedupeBlockAccounts.SumNanoseconds + dst.SourceBatch.Count += src.SourceBatch.Count + dst.SourceBatch.SumNanoseconds += src.SourceBatch.SumNanoseconds + dst.ParentMapBuild.Count += src.ParentMapBuild.Count + dst.ParentMapBuild.SumNanoseconds += src.ParentMapBuild.SumNanoseconds + dst.SysvarUpdates.Count += src.SysvarUpdates.Count + dst.SysvarUpdates.SumNanoseconds += src.SysvarUpdates.SumNanoseconds + dst.SysvarClockRead.Count += src.SysvarClockRead.Count + dst.SysvarClockRead.SumNanoseconds += src.SysvarClockRead.SumNanoseconds + dst.SysvarSlotHashesRead.Count += src.SysvarSlotHashesRead.Count + dst.SysvarSlotHashesRead.SumNanoseconds += src.SysvarSlotHashesRead.SumNanoseconds + dst.SysvarRecentBlockhashesRead.Count += src.SysvarRecentBlockhashesRead.Count + dst.SysvarRecentBlockhashesRead.SumNanoseconds += src.SysvarRecentBlockhashesRead.SumNanoseconds + dst.SysvarSlotHistoryRead.Count += src.SysvarSlotHistoryRead.Count + dst.SysvarSlotHistoryRead.SumNanoseconds += src.SysvarSlotHistoryRead.SumNanoseconds + dst.SysvarStakeHistoryRead.Count += src.SysvarStakeHistoryRead.Count + dst.SysvarStakeHistoryRead.SumNanoseconds += src.SysvarStakeHistoryRead.SumNanoseconds + dst.SysvarLastRestartSlotRead.Count += src.SysvarLastRestartSlotRead.Count + dst.SysvarLastRestartSlotRead.SumNanoseconds += src.SysvarLastRestartSlotRead.SumNanoseconds + dst.WorkingSetLookup.Count += src.WorkingSetLookup.Count + dst.WorkingSetLookup.SumNanoseconds += src.WorkingSetLookup.SumNanoseconds + dst.InProgressLookup.Count += src.InProgressLookup.Count + dst.InProgressLookup.SumNanoseconds += src.InProgressLookup.SumNanoseconds + dst.AppendVecPinWait.Count += src.AppendVecPinWait.Count + dst.AppendVecPinWait.SumNanoseconds += src.AppendVecPinWait.SumNanoseconds + dst.ReadCacheEpochWait.Count += src.ReadCacheEpochWait.Count + dst.ReadCacheEpochWait.SumNanoseconds += src.ReadCacheEpochWait.SumNanoseconds + dst.CacheLookup.Count += src.CacheLookup.Count + dst.CacheLookup.SumNanoseconds += src.CacheLookup.SumNanoseconds + dst.AdmissionFilter.Count += src.AdmissionFilter.Count + dst.AdmissionFilter.SumNanoseconds += src.AdmissionFilter.SumNanoseconds + dst.IndexLookup.Count += src.IndexLookup.Count + dst.IndexLookup.SumNanoseconds += src.IndexLookup.SumNanoseconds + dst.ReadPlanning.Count += src.ReadPlanning.Count + dst.ReadPlanning.SumNanoseconds += src.ReadPlanning.SumNanoseconds + dst.AppendVecRead.Count += src.AppendVecRead.Count + dst.AppendVecRead.SumNanoseconds += src.AppendVecRead.SumNanoseconds + dst.CachePublicationWait.Count += src.CachePublicationWait.Count + dst.CachePublicationWait.SumNanoseconds += src.CachePublicationWait.SumNanoseconds + dst.CachePublication.Count += src.CachePublication.Count + dst.CachePublication.SumNanoseconds += src.CachePublication.SumNanoseconds + dst.SysvarWorkingSetLookup.Count += src.SysvarWorkingSetLookup.Count + dst.SysvarWorkingSetLookup.SumNanoseconds += src.SysvarWorkingSetLookup.SumNanoseconds + dst.SysvarClone.Count += src.SysvarClone.Count + dst.SysvarClone.SumNanoseconds += src.SysvarClone.SumNanoseconds + dst.SysvarAppendVecPinWait.Count += src.SysvarAppendVecPinWait.Count + dst.SysvarAppendVecPinWait.SumNanoseconds += src.SysvarAppendVecPinWait.SumNanoseconds + dst.SysvarInProgressLookup.Count += src.SysvarInProgressLookup.Count + dst.SysvarInProgressLookup.SumNanoseconds += src.SysvarInProgressLookup.SumNanoseconds + dst.SysvarReadCacheEpochWait.Count += src.SysvarReadCacheEpochWait.Count + dst.SysvarReadCacheEpochWait.SumNanoseconds += src.SysvarReadCacheEpochWait.SumNanoseconds + dst.SysvarCacheLookup.Count += src.SysvarCacheLookup.Count + dst.SysvarCacheLookup.SumNanoseconds += src.SysvarCacheLookup.SumNanoseconds + dst.SysvarIndexAndAppendVecRead.Count += src.SysvarIndexAndAppendVecRead.Count + dst.SysvarIndexAndAppendVecRead.SumNanoseconds += src.SysvarIndexAndAppendVecRead.SumNanoseconds + dst.SysvarCachePublicationWait.Count += src.SysvarCachePublicationWait.Count + dst.SysvarCachePublicationWait.SumNanoseconds += src.SysvarCachePublicationWait.SumNanoseconds + dst.SysvarCachePublication.Count += src.SysvarCachePublication.Count + dst.SysvarCachePublication.SumNanoseconds += src.SysvarCachePublication.SumNanoseconds + dst.RequestedKeys += src.RequestedKeys + dst.DurableKeys += src.DurableKeys + dst.ParentAccounts += src.ParentAccounts + dst.WorkingSetHits += src.WorkingSetHits + dst.InProgressHits += src.InProgressHits + dst.PendingFoldHits += src.PendingFoldHits + dst.CacheHits += src.CacheHits + dst.IndexHits += src.IndexHits + dst.IndexMisses += src.IndexMisses + dst.UniqueAppendVecs += src.UniqueAppendVecs + dst.AppendVecChunks += src.AppendVecChunks + dst.AppendVecAccounts += src.AppendVecAccounts + dst.OpenFailures += src.OpenFailures + dst.ReadFailures += src.ReadFailures + dst.RetryAccounts += src.RetryAccounts + dst.CommonCacheAdmissions += src.CommonCacheAdmissions + dst.CommonCacheAdmissionsSkipped += src.CommonCacheAdmissionsSkipped + dst.VoteCacheAdmissions += src.VoteCacheAdmissions + dst.VoteCacheAdmissionsSkipped += src.VoteCacheAdmissionsSkipped + dst.CachePublicationEpochRejects += src.CachePublicationEpochRejects + dst.DecodedAccountObjects += src.DecodedAccountObjects + dst.DecodedAccountBytes += src.DecodedAccountBytes + dst.PlaceholderObjects += src.PlaceholderObjects + dst.SysvarReads += src.SysvarReads + dst.SysvarWorkingSetHits += src.SysvarWorkingSetHits + dst.SysvarInProgressHits += src.SysvarInProgressHits + dst.SysvarPendingFoldHits += src.SysvarPendingFoldHits + dst.SysvarCacheHits += src.SysvarCacheHits + dst.SysvarDurableReads += src.SysvarDurableReads + dst.SysvarCachePublicationEpochRejects += src.SysvarCachePublicationEpochRejects +} + +// TurbineIngress records per-slot pre-replay pipeline observations. // It is written to replay_timings.jsonl without high-cardinality metric labels. type TurbineIngress struct { ShredCollection Timing CompletionQueueDelay Timing BlockDecode Timing + // Completion-only parse and outstanding-signature join/verification time. TransactionParse Timing TransactionSigverify Timing ReplayAdmission Timing + // Summed completed prefetched component durations, including any discarded + // optimistic prefix. Overlap reception; not CPU time or additive wall stages. + // Early sigverify includes queueing. + EarlyTransactionParse Timing + EarlyTransactionSigverify Timing + // Completion wait for claimed background parsing/submission, outside BlockDecode. + EarlyPreparationWait Timing + EarlyVerifiedTransactions uint64 + // FullToReady contains the completion stages above, excluding admission. + FullToReady Timing } // VoteRewardDetails decomposes RewardCertificatePreflight and @@ -124,6 +244,119 @@ type VoteRewardDetails struct { VoteAccountsUpdated uint64 } +// StreamingExecution records execution that ran while a block's shreds were +// still arriving. Groups are the contiguous ready-batch sets executed per +// wake-up; Transactions counts what they executed. TxLoopBeforeFull is the +// group execution wall time that finished before the slot was fully +// assembled, i.e. the work hidden behind reception. OpenDelay runs from the +// header batch being decoded to the stream opening; the timeline fields and +// the OpenWait* timings below say what held the child (its parent's arrival, +// its parent's replay, or the loop itself). Discarded is 1 when a stream for +// this slot was thrown away and the block was executed whole; DiscardReason +// names why. +type StreamingExecution struct { + // VerificationWait is wall time joining speculative verification groups, + // including failed joins. It overlaps GroupJoinAssembly for successful + // groups; it is neither crypto CPU time nor additional replay latency. + VerificationWait Timing + Opened uint64 + Groups uint64 + Transactions uint64 + TxLoopBeforeFull Timing + OpenDelay Timing + Discarded uint64 + DiscardReason string + + // Timeline: wall-clock unix nanoseconds of the events that bound the + // stream's open, zero when unknown. They join with the parent's record + // (ParentFullNanos is the parent's FullNanos) and with external captures. + // + // HeaderReadyNanos the child's header batch was decoded + // HeaderSeenNanos the executor first handled that header (from its + // wake-up, or recovered from the assembler after a + // dropped wake-up, in which case ready is the + // lookup instant and seen follows it at once) + // ParentFullNanos the parent's last shred (0: parent was a skip or + // not a turbine block) + // ParentAdmittedNanos the source handed the parent to replay (its own + // ancestors replayed, the emitter released it) + // ParentReplayedNanos the parent's replay result reached consensus and + // the frontier advanced to it (0: unknown, e.g. the + // frontier was re-based by a fork switch) + // OpenedNanos the stream's bank opened + // FirstGroupStartNanos the first executed group started + // WaitEnteredNanos the loop first entered the replay wait after the + // parent was replayed (0: unknown) + // FullNanos this block's last shred + // FinalizeStartNanos the complete block reached the stream + HeaderReadyNanos int64 + HeaderSeenNanos int64 + ParentFullNanos int64 + ParentAdmittedNanos int64 + ParentReplayedNanos int64 + WaitEnteredNanos int64 + OpenedNanos int64 + FirstGroupStartNanos int64 + FullNanos int64 + FinalizeStartNanos int64 + + // OpenDelay decomposed into attributable waits (each zero when the + // timeline cannot support it): + // OpenWaitParentArrival header ready → parent's last shred: the child's + // header was decoded before its parent was even + // fully received (arrival timing; cause not established) + // OpenWaitParentReplay parent's last shred (or header ready, whichever + // is later) → parent replayed: the parent's own + // post-full path held the child; split, when the + // parent's admission is known, into + // OpenWaitParentPreAdmission … → admission, including any streamed execution + // OpenWaitParentPostAdmission admission → replayed, including finalization + // OpenWaitLoop max(parent replayed, header ready) → opened; + // includes time before the header is handled + // OpenWaitPostReplay loop start → first wait entry after the parent + // OpenWaitDispatch max(wait entry, loop start) → opened + // These are elapsed intervals, not CPU times. Subdivisions must not be + // added to their aggregates. Unknown parent milestones limit attribution. + OpenWaitParentArrival Timing + OpenWaitParentReplay Timing + OpenWaitParentPreAdmission Timing + OpenWaitParentPostAdmission Timing + OpenWaitLoop Timing + OpenWaitPostReplay Timing + OpenWaitDispatch Timing + + // Groups, from the executor's per-group bookkeeping (each group is the + // contiguous set of decoded batches that was ready at one wake-up, or + // the finalize suffix): + // GroupJoinAssembly verification joins plus batch-slice assembly; + // not pure cryptography or verifier queue time + // GroupJoinAssemblyAfterFull intersection of that interval with post-full + // GroupPreparation identity binding and execution-copy creation + // GroupPreparationAfterFull intersection of preparation with post-full + // TxLoopAfterFull group/suffix execution after the last shred: + // the execution FullToReplayed actually paid for + // GroupsStraddlingFull groups that started before and finished after + // LargestGroupTransactions / LargestGroupBatches: the biggest group (a + // late open turns the whole backlog into one); + // Batches is 0 when that group was the suffix + // LastGroupEndNanos when the last group (or suffix) finished + GroupJoinAssembly Timing + GroupJoinAssemblyAfterFull Timing + GroupPreparation Timing + GroupPreparationAfterFull Timing + TxLoopAfterFull Timing + GroupsStraddlingFull uint64 + LargestGroupTransactions uint64 + LargestGroupBatches uint64 + LastGroupEndNanos int64 + + // NotOpenedReason is set when the block was executed whole without a + // stream having opened for it: why the executor never opened one + // ("header_not_seen", "declined:", "waiting_for_parent:…"). + // Empty when a stream opened (see Discarded for the ones thrown away). + NotOpenedReason string +} + // Metrics for replaying a single block type BlockReplay struct { Slot uint64 @@ -171,16 +404,20 @@ type BlockReplay struct { // BlockUpdateAccounts is synchronous critical-path work: rooted-tail // buffering (including its callback) or legacy store enqueue. It excludes // legacy asynchronous disk completion. - BlockUpdateAccounts Timing - TransactionStatusCommit Timing - SignatureVerificationJoin Timing - AccountsDeltaHash Timing - LtHashDedupe Timing - LtHashWorkerCompute Timing - LtHashPartialReduce Timing - BankHashFinalize Timing - BankHash Timing - AlpenglowFooterVerification Timing + BlockUpdateAccounts Timing + TransactionStatusCommit Timing + // Preparation overlaps execution and is not additive with replay wall time. + // PreparationWait is the residual join nested within TransactionStatusCommit. + TransactionStatusPreparation Timing + TransactionStatusPreparationWait Timing + SignatureVerificationJoin Timing + AccountsDeltaHash Timing + LtHashDedupe Timing + LtHashWorkerCompute Timing + LtHashPartialReduce Timing + BankHashFinalize Timing + BankHash Timing + AlpenglowFooterVerification Timing // PostProcessBlock is caller-side state publication and replay // bookkeeping after ProcessBlock returns. TransactionStatusView, // ChainTipUpdate, and ResumeContext are nested sub-phases; logging, summary @@ -190,6 +427,15 @@ type BlockReplay struct { ChainTipUpdate Timing ResumeContext Timing + // FullToReplayed is the vote-path latency Mithril controls: wall time from + // the last shred of a turbine block being assembled (the assembler's fullAt) + // to the replay result being handed to consensus. Absent for blocks that + // did not arrive as shreds. It is the number streaming execution reduces. + FullToReplayed Timing + // StreamingExecution summarizes any execution that overlapped shred + // reception for this block; all zero when the block was executed whole. + StreamingExecution StreamingExecution + LtHashInputAccounts uint64 LtHashUniqueAccounts uint64 LtHashUnchangedAccounts uint64 diff --git a/pkg/replay/account_loader_metrics_test.go b/pkg/replay/account_loader_metrics_test.go new file mode 100644 index 000000000..128bd1cef --- /dev/null +++ b/pkg/replay/account_loader_metrics_test.go @@ -0,0 +1,34 @@ +package replay + +import ( + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/stretchr/testify/require" +) + +func TestAccountLoaderRetainsGroupsAcrossReset(t *testing.T) { + previous := metrics.GlobalBlockReplay.AccountLoader + t.Cleanup(func() { metrics.GlobalBlockReplay.AccountLoader = previous }) + exec := &blockExecution{} + metrics.GlobalBlockReplay.AccountLoader = metrics.AccountLoader{RequestedKeys: 99} + func() { + defer exec.captureAccountLoader()() + recordAccountLoaderBatchStats(&metrics.GlobalBlockReplay.AccountLoader, accountsdb.BatchReadStats{RequestedKeys: 3, IndexHits: 2, AppendVecAccounts: 2, AppendVecReadNanoseconds: 1000}) + recordAccountLoaderBatchStats(&metrics.GlobalBlockReplay.AccountLoader, accountsdb.BatchReadStats{RequestedKeys: 4, CacheHits: 4}) + }() + require.EqualValues(t, 99, metrics.GlobalBlockReplay.AccountLoader.RequestedKeys, "another slot's record is unchanged") + metrics.GlobalBlockReplay.AccountLoader = metrics.AccountLoader{} + func() { + defer exec.captureAccountLoader()() + recordAccountLoaderBatchStats(&metrics.GlobalBlockReplay.AccountLoader, accountsdb.BatchReadStats{RequestedKeys: 5, CacheHits: 5}) + }() + require.Zero(t, metrics.GlobalBlockReplay.AccountLoader.RequestedKeys, "speculative work is not published before acceptance") + require.EqualValues(t, 12, exec.accountLoader.RequestedKeys) + require.EqualValues(t, 9, exec.accountLoader.CacheHits) + require.EqualValues(t, 2, exec.accountLoader.AppendVecAccounts) + require.EqualValues(t, time.Microsecond, exec.accountLoader.AppendVecRead.SumNanoseconds) + require.EqualValues(t, 3, exec.accountLoader.AppendVecRead.Count) +} diff --git a/pkg/replay/alpenglow_switch.go b/pkg/replay/alpenglow_switch.go index 76f557971..2d384a096 100644 --- a/pkg/replay/alpenglow_switch.go +++ b/pkg/replay/alpenglow_switch.go @@ -13,6 +13,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/rewards" "github.com/Overclock-Validator/mithril/pkg/sealevel" "github.com/Overclock-Validator/mithril/pkg/state" + "github.com/Overclock-Validator/mithril/pkg/turbine" "github.com/gagliardetto/solana-go" ) @@ -116,6 +117,16 @@ func newAlpenglowSwitchSweeper(engine consensusengine.Engine) *alpenglowSwitchSw return s } +// peek tests for a switch without consuming the sweep's version/frontier gate. +// Streaming admission must leave a detected switch for the replay loop to apply. +func (s *alpenglowSwitchSweeper) peek(executed map[uint64]solana.Hash, lastRooted, tip uint64) *CertifiedSwitch { + if s == nil { + return nil + } + snapshot := *s + return snapshot.sweep(executed, lastRooted, tip) +} + // sweep walks consumed block/skip outcomes in (lastRooted, tip] and returns // the first contradiction with a decisive chain decision. tip includes trailing // skips even when the executed bank remains at an earlier slot. @@ -175,23 +186,86 @@ func waitForAlpenglowReplayInput( decisionChanges <-chan struct{}, pollInterval time.Duration, ) (*b.Block, *blockstream.AlpenglowParentSwitch, *CertifiedSwitch) { + adapted := func(ctx context.Context, decisionChanges <-chan struct{}, _ <-chan turbine.StreamEvent, _ <-chan time.Time) blockstream.ReplayInput { + block, parentSwitch, decisionChanged := next(ctx, decisionChanges) + return blockstream.ReplayInput{Block: block, ParentSwitch: parentSwitch, DecisionChanged: decisionChanged} + } + return waitForReplayInput(ctx, adapted, sweep, decisionChanges, pollInterval, nil) +} + +// replayStreamer is the streaming executor as the wait loop drives it: the +// feed it listens to, its poll timer (nil while idle), and the two handlers, +// which execute transaction groups on the replay goroutine. +type replayStreamer interface { + events() <-chan turbine.StreamEvent + tick() <-chan time.Time + handleEvent(turbine.StreamEvent) + handleTick() +} + +// replayInputSource is BlockSource.NextReplayInput. +type replayInputSource func(ctx context.Context, decisionChanges <-chan struct{}, streamEvents <-chan turbine.StreamEvent, streamTick <-chan time.Time) blockstream.ReplayInput + +// waitForReplayInput is waitForAlpenglowReplayInput with the streaming +// executor folded into the same wait: feed wake-ups and poll ticks are +// handled here, on the replay goroutine, and never end the wait, so the loop +// body only ever sees a block, a switch, or an ended wait exactly as before. +// streamer may be nil (no streaming); sweep may be nil (no certificate +// correction), in which case decision notifications stay disabled and the +// wait blocks without polling, as it always did. +func waitForReplayInput( + ctx context.Context, + next replayInputSource, + sweep func() *CertifiedSwitch, + decisionChanges <-chan struct{}, + pollInterval time.Duration, + streamer replayStreamer, +) (*b.Block, *blockstream.AlpenglowParentSwitch, *CertifiedSwitch) { + var streamEvents <-chan turbine.StreamEvent + if streamer != nil { + // A slot may have become eligible while the previous block executed + // (its header arrived mid-execution); open it before blocking. + streamer.handleTick() + streamEvents = streamer.events() + } if sweep == nil { - block, parentSwitch, _ := next(ctx, nil) - return block, parentSwitch, nil + decisionChanges = nil + if streamer == nil { + in := next(ctx, nil, nil, nil) + return in.Block, in.ParentSwitch, nil + } } for { if ctx.Err() != nil { return nil, nil, nil } - if sw := sweep(); sw != nil { - return nil, nil, sw + if sweep != nil { + if sw := sweep(); sw != nil { + return nil, nil, sw + } + } + waitCtx, cancel := ctx, func() {} + if sweep != nil { + waitCtx, cancel = context.WithTimeout(ctx, pollInterval) } - waitCtx, cancel := context.WithTimeout(ctx, pollInterval) - block, parentSwitch, decisionChanged := next(waitCtx, decisionChanges) - timedOut := waitCtx.Err() == context.DeadlineExceeded + var tick <-chan time.Time + if streamer != nil { + tick = streamer.tick() + } + in := next(waitCtx, decisionChanges, streamEvents, tick) + timedOut := sweep != nil && waitCtx.Err() == context.DeadlineExceeded cancel() - if block != nil || parentSwitch != nil || (!decisionChanged && !timedOut) { - return block, parentSwitch, nil + switch { + case in.Block != nil || in.ParentSwitch != nil: + return in.Block, in.ParentSwitch, nil + case in.StreamEvent != nil: + streamer.handleEvent(*in.StreamEvent) + case in.StreamTick: + streamer.handleTick() + case in.DecisionChanged || timedOut: + // Re-sweep, then wait again. + default: + return nil, nil, nil } } } diff --git a/pkg/replay/alpenglow_switch_test.go b/pkg/replay/alpenglow_switch_test.go index cd8084682..35cd7ca17 100644 --- a/pkg/replay/alpenglow_switch_test.go +++ b/pkg/replay/alpenglow_switch_test.go @@ -455,3 +455,16 @@ func TestWaitForAlpenglowReplayInputHonorsCancellation(t *testing.T) { require.Nil(t, parentSwitch) require.Nil(t, certifiedSwitch) } + +func TestSwitchPeekDoesNotConsumeDecision(t *testing.T) { + q := &fakeChainQuery{certified: map[uint64]alpenglow.BlockID{101: {Slot: 101, Hash: swHash(9)}}, skipped: map[uint64]bool{}, version: 1} + s := newTestSweeper(q) + executed := map[uint64]solana.Hash{101: swHash(1)} + before := *s + first := s.peek(executed, 100, 101) + require.NotNil(t, first) + require.Equal(t, before, *s) + require.Equal(t, first, s.peek(executed, 100, 101)) + require.Equal(t, first, s.sweep(executed, 100, 101), "replay still receives the switch") + require.Nil(t, s.sweep(executed, 100, 101), "normal sweep still consumes its gate") +} diff --git a/pkg/replay/alpenglow_unwind_test.go b/pkg/replay/alpenglow_unwind_test.go index dd478e0fc..52e949c2b 100644 --- a/pkg/replay/alpenglow_unwind_test.go +++ b/pkg/replay/alpenglow_unwind_test.go @@ -7,10 +7,12 @@ import ( "testing" "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/global" "github.com/Overclock-Validator/mithril/pkg/rewards" "github.com/Overclock-Validator/mithril/pkg/sealevel" "github.com/Overclock-Validator/mithril/pkg/state" bin "github.com/gagliardetto/binary" + "github.com/gagliardetto/solana-go" "github.com/mr-tron/base58" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -373,3 +375,25 @@ func assertUnwindFallbackReason(t *testing.T, want string, sw *CertifiedSwitch, assert.Nil(t, bankSysvars) assert.Equal(t, want, reason) } + +func TestUnwindPreservesPendingLeaderStakeEntries(t *testing.T) { + tail := newUnrootedTail(&fakeDurable{}, &fakeCommitter{durable: accounts.NewMemAccounts()}, 512, 1, "") + for _, slot := range []uint64{8, 9, 12} { + global.EnqueuePendingStakePubkey(slot, solana.PublicKey{byte(slot), 0xFA}) + t.Cleanup(func() { global.DropPendingStakePubkeys(slot) }) + } + for _, slot := range []uint64{8, 9} { + tail.Add(slot, []*accounts.Account{testAccount(1, slot)}, testHashBytes(byte(slot))) + } + tail.unwind(8) + entries := global.PendingStakeEntriesSnapshot() + for _, slot := range []uint64{8, 9, 12} { + found := false + for _, entry := range entries { + if entry.Pubkey == (solana.PublicKey{byte(slot), 0xFA}) { + found = true + } + } + require.Equal(t, slot == 12, found) + } +} diff --git a/pkg/replay/async_checkpoint_capture_test.go b/pkg/replay/async_checkpoint_capture_test.go new file mode 100644 index 000000000..a556c5a37 --- /dev/null +++ b/pkg/replay/async_checkpoint_capture_test.go @@ -0,0 +1,172 @@ +package replay + +import ( + "encoding/json" + "errors" + "sync" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/state" + "github.com/stretchr/testify/require" +) + +type testCheckpointEncoder func() ([]byte, error) + +func (f testCheckpointEncoder) MarshalBinary() ([]byte, error) { return f() } + +func testCheckpointBytes(payload []byte) TransactionStatusSnapshot { + owned := append([]byte(nil), payload...) + return testCheckpointEncoder(func() ([]byte, error) { return append([]byte(nil), owned...), nil }) +} + +func TestAsyncCheckpointEncodingDoesNotRunDuringJobBuild(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts()} + tail := asyncTestTail(fc, 5, 6) + started, release := make(chan struct{}), make(chan struct{}) + blockEncoding := false + rootDir := t.TempDir() + require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ + Capture: func(through uint64) (TransactionStatusSnapshot, error) { + require.Equal(t, uint64(6), through) + return testCheckpointEncoder(func() ([]byte, error) { + if !blockEncoding { + return nil, errors.New("encoder ran during job construction") + } + close(started) + <-release + return []byte("encoded-on-worker"), nil + }), nil + }, + Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { + return PrepareTransactionStatusCheckpoint(rootDir, through, payload) + }, + })) + job, err := tail.buildFoldJob(6, false) + require.NoError(t, err) + select { + case <-started: + t.Fatal("job construction ran the encoder") + default: + } + promoter := newAsyncPromoter(fc) + // Always unblock the worker before draining it, including a failed assertion. + var releaseOnce sync.Once + unblock := func() { releaseOnce.Do(func() { close(release) }) } + defer promoter.stop() + defer unblock() + blockEncoding = true + promoter.enqueue(job) + select { + case <-started: + case <-time.After(2 * time.Second): + t.Fatal("checkpoint worker did not start encoding") + } + // Replay can publish a later bank while the worker's encoder is blocked. + tail.Add(7, []*accounts.Account{testAccount(3, 7)}, testHashBytes(7)) + tail.SetContext(7, &state.ResumeContext{Slot: 7}) + require.Equal(t, 3, tail.overlay.HeldSlots()) + require.Nil(t, promoter.poll()) + + unblock() + result := promoter.drain() + require.NotNil(t, result) + require.NoError(t, result.err) + require.Nil(t, result.job.transactionStatusSnapshot) + tail.applyFoldJob(result.job) + require.Equal(t, 1, tail.overlay.HeldSlots()) +} + +func TestCheckpointCaptureFailureOrdering(t *testing.T) { + cases := []struct { + name string + capture func(uint64) (TransactionStatusSnapshot, error) + want string + buildFails bool + }{ + {"nil capture", func(uint64) (TransactionStatusSnapshot, error) { return nil, nil }, "capture is nil", true}, + {"capture error", func(uint64) (TransactionStatusSnapshot, error) { return nil, errors.New("capture failed") }, "capture failed", true}, + {"encode error", func(uint64) (TransactionStatusSnapshot, error) { + return testCheckpointEncoder(func() ([]byte, error) { return nil, errors.New("encode failed") }), nil + }, "encode failed", false}, + {"empty encoding", func(uint64) (TransactionStatusSnapshot, error) { return testCheckpointBytes(nil), nil }, "snapshot is empty", false}, + } + for _, tc := range cases { + for _, forced := range []bool{false, true} { + name := tc.name + "/async" + if forced { + name = tc.name + "/forced" + } + t.Run(name, func(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts()} + tail := asyncTestTail(fc, 5, 6) + installed := false + require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ + Capture: tc.capture, + Install: func(uint64, []byte) (*state.TransactionStatusCheckpointRef, error) { + installed = true + return nil, errors.New("unexpected install") + }, + })) + if forced { + through, _, err := tail.flush(6) + require.ErrorContains(t, err, tc.want) + require.Zero(t, through) + } else { + job, err := tail.buildFoldJob(6, false) + if tc.buildFails { + require.ErrorContains(t, err, tc.want) + require.Nil(t, job) + } else { + require.NoError(t, err) + require.ErrorContains(t, runFoldJob(fc, job), tc.want) + require.Nil(t, job.transactionStatusSnapshot, "failed result retained its captured deltas") + } + } + require.False(t, installed) + require.Empty(t, fc.throughs) + require.Equal(t, 2, tail.overlay.HeldSlots()) + }) + } + } +} + +func TestFoldCheckpointKeepsCapturedRootAfterLiveCacheAdvances(t *testing.T) { + c := importedStatusCacheForTest(t) + for slot := uint64(301); slot <= 350; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + } + want, err := legacyStatusSnapshotForTest(c, 320) + require.NoError(t, err) + fc := &fakeCommitter{durable: accounts.NewMemAccounts()} + tail := asyncTestTail(fc, 319, 320, 321) + rootDir := t.TempDir() + require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ + Capture: c.CaptureSnapshotThrough, + Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { + return PrepareTransactionStatusCheckpoint(rootDir, through, payload) + }, + })) + job, err := tail.buildFoldJob(321, false) + require.NoError(t, err) + for slot := uint64(351); slot <= 660; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + c.Root(slot - 20) + } + require.NoError(t, runFoldJob(fc, job)) + require.Nil(t, job.transactionStatusSnapshot, "completed result retained captured deltas") + require.Positive(t, job.checkpointCaptureTime) + require.Positive(t, job.checkpointEncodeTime) + require.Equal(t, len(want), job.checkpointBytes) + var manifest state.ResumeContext + require.NoError(t, json.Unmarshal(fc.ctxs[320], &manifest)) + require.Equal(t, uint64(320), manifest.TransactionStatusCheckpoint.Root) + got, err := ReadTransactionStatusCheckpoint(rootDir, manifest.TransactionStatusCheckpoint) + require.NoError(t, err) + require.Equal(t, want, got) + restored, err := NewTransactionStatusCacheFromSnapshot(got) + require.NoError(t, err) + require.Equal(t, uint64(320), restored.RootedThrough()) + require.True(t, restored.CoverageComplete()) +} diff --git a/pkg/replay/async_promotion_test.go b/pkg/replay/async_promotion_test.go index 49d47a49e..2780e425d 100644 --- a/pkg/replay/async_promotion_test.go +++ b/pkg/replay/async_promotion_test.go @@ -22,7 +22,7 @@ type slowCommitter struct { delay time.Duration } -func TestFoldJobSnapshotsStatusOnLoopAndReferenceRidesManifest(t *testing.T) { +func TestFoldJobCapturesStatusOnLoopAndReferenceRidesManifest(t *testing.T) { rootDir := t.TempDir() fc := &fakeCommitter{durable: accounts.NewMemAccounts()} tail := asyncTestTail(fc, 5, 6, 7) @@ -32,10 +32,10 @@ func TestFoldJobSnapshotsStatusOnLoopAndReferenceRidesManifest(t *testing.T) { installCalled := false afterCommitCalled := false hooks := TransactionStatusCheckpointHooks{ - Snapshot: func(through uint64) ([]byte, error) { + Capture: func(through uint64) (TransactionStatusSnapshot, error) { require.Equal(t, uint64(6), through) snapshotCalled = true - return scratch, nil + return testCheckpointBytes(scratch), nil }, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { require.True(t, snapshotCalled, "worker install ran before loop snapshot") @@ -83,7 +83,7 @@ func TestFoldJobCheckpointFailuresCannotReachCommitBatch(t *testing.T) { fc := &fakeCommitter{durable: accounts.NewMemAccounts()} tail := asyncTestTail(fc, 5, 6) require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(uint64) ([]byte, error) { return nil, errors.New("snapshot boom") }, + Capture: func(uint64) (TransactionStatusSnapshot, error) { return nil, errors.New("snapshot boom") }, Install: func(uint64, []byte) (*state.TransactionStatusCheckpointRef, error) { t.Fatal("install must not run") return nil, nil @@ -101,7 +101,7 @@ func TestFoldJobCheckpointFailuresCannotReachCommitBatch(t *testing.T) { fc := &fakeCommitter{durable: accounts.NewMemAccounts()} tail := asyncTestTail(fc, 5, 6) require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(uint64) ([]byte, error) { return []byte("captured"), nil }, + Capture: func(uint64) (TransactionStatusSnapshot, error) { return testCheckpointBytes([]byte("captured")), nil }, Install: func(uint64, []byte) (*state.TransactionStatusCheckpointRef, error) { return nil, errors.New("fsync boom") }, @@ -120,7 +120,7 @@ func TestFoldJobCheckpointFailuresCannotReachCommitBatch(t *testing.T) { tail := asyncTestTail(fc, 5, 6) afterCommitCalled := false require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(uint64) ([]byte, error) { return []byte("captured"), nil }, + Capture: func(uint64) (TransactionStatusSnapshot, error) { return testCheckpointBytes([]byte("captured")), nil }, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { return PrepareTransactionStatusCheckpoint(rootDir, through, payload) }, @@ -144,9 +144,9 @@ func TestForcedFoldCarriesStatusCheckpointReference(t *testing.T) { tail := asyncTestTail(fc, 5) afterCommitCalled := false require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(through uint64) ([]byte, error) { + Capture: func(through uint64) (TransactionStatusSnapshot, error) { require.Equal(t, uint64(5), through) - return []byte("forced-partial-status"), nil + return testCheckpointBytes([]byte("forced-partial-status")), nil }, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { return PrepareTransactionStatusCheckpoint(rootDir, through, payload) @@ -175,7 +175,7 @@ func TestCheckpointAfterCommitRequiresDurabilityHooks(t *testing.T) { err := tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ AfterCommit: func(*state.TransactionStatusCheckpointRef) error { return nil }, }) - require.ErrorContains(t, err, "requires Snapshot and Install") + require.ErrorContains(t, err, "requires Capture and Install") } func (c *slowCommitter) CommitBatch(deltas []accounts.SlotDelta, throughSlot uint64, bankhashes map[uint64][32]byte, resumeCtx []byte) (accountsdb.BatchCommitResult, error) { @@ -444,3 +444,26 @@ func TestShutdownFlushCannotFoldPastGateTarget(t *testing.T) { target = safePromoteTarget(9, true, 7, 6) assert.Equal(t, uint64(5), target, "persisted-divergence floor holds promotion below the disputed slot") } + +// Model repeated replay/skip iterations while a nearly full checkpoint batch +// waits for one more held bank. Account writes must not be copied on this path. +func BenchmarkBuildFoldJobWaitingForBatch(b *testing.B) { + tail := newUnrootedTail(&fakeDurable{}, &fakeCommitter{durable: accounts.NewMemAccounts()}, 512, 128, "") + writes := make([]*accounts.Account, 512) + for i := range writes { + var key [32]byte + key[0], key[1] = byte(i), byte(i>>8) + writes[i] = &accounts.Account{Key: key, Lamports: 1} + } + for slot := uint64(1); slot <= 127; slot++ { + tail.Add(slot, writes, nil) + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + job, err := tail.buildFoldJob(127, false) + if err != nil || job != nil { + b.Fatalf("unexpected fold admission: job=%v err=%v", job, err) + } + } +} diff --git a/pkg/replay/block.go b/pkg/replay/block.go index 495ed53a8..e571c98d0 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -25,7 +25,6 @@ import ( "github.com/Overclock-Validator/mithril/pkg/accountsdb" a "github.com/Overclock-Validator/mithril/pkg/addresses" "github.com/Overclock-Validator/mithril/pkg/arena" - "github.com/Overclock-Validator/mithril/pkg/bankhash" "github.com/Overclock-Validator/mithril/pkg/base58" b "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/blockstream" @@ -37,7 +36,6 @@ import ( "github.com/Overclock-Validator/mithril/pkg/lthash" "github.com/Overclock-Validator/mithril/pkg/metrics" "github.com/Overclock-Validator/mithril/pkg/mlog" - "github.com/Overclock-Validator/mithril/pkg/rent" "github.com/Overclock-Validator/mithril/pkg/rewards" "github.com/Overclock-Validator/mithril/pkg/rpcclient" "github.com/Overclock-Validator/mithril/pkg/sealevel" @@ -1032,28 +1030,28 @@ func loadBlockAccountsAndUpdateSysvars( } func recordAccountLoaderBatchStats(dst *metrics.AccountLoader, src accountsdb.BatchReadStats) { - dst.RequestedKeys = src.RequestedKeys - dst.DurableKeys = src.DurableKeys - dst.WorkingSetHits = src.WorkingSetHits - dst.InProgressHits = src.InProgressHits - dst.PendingFoldHits = src.PendingFoldHits - dst.CacheHits = src.CacheHits - dst.IndexHits = src.IndexHits - dst.IndexMisses = src.IndexMisses - dst.UniqueAppendVecs = src.UniqueAppendVecs - dst.AppendVecChunks = src.AppendVecChunks - dst.AppendVecAccounts = src.AppendVecAccounts - dst.OpenFailures = src.OpenFailures - dst.ReadFailures = src.ReadFailures - dst.RetryAccounts = src.RetryAccounts - dst.CommonCacheAdmissions = src.CommonCacheAdmissions - dst.CommonCacheAdmissionsSkipped = src.CommonCacheAdmissionsSkipped - dst.VoteCacheAdmissions = src.VoteCacheAdmissions - dst.VoteCacheAdmissionsSkipped = src.VoteCacheAdmissionsSkipped - dst.CachePublicationEpochRejects = src.CachePublicationEpochRejects - dst.DecodedAccountObjects = src.DecodedAccountObjects - dst.DecodedAccountBytes = src.DecodedAccountBytes - dst.PlaceholderObjects = src.PlaceholderObjects + dst.RequestedKeys += src.RequestedKeys + dst.DurableKeys += src.DurableKeys + dst.WorkingSetHits += src.WorkingSetHits + dst.InProgressHits += src.InProgressHits + dst.PendingFoldHits += src.PendingFoldHits + dst.CacheHits += src.CacheHits + dst.IndexHits += src.IndexHits + dst.IndexMisses += src.IndexMisses + dst.UniqueAppendVecs += src.UniqueAppendVecs + dst.AppendVecChunks += src.AppendVecChunks + dst.AppendVecAccounts += src.AppendVecAccounts + dst.OpenFailures += src.OpenFailures + dst.ReadFailures += src.ReadFailures + dst.RetryAccounts += src.RetryAccounts + dst.CommonCacheAdmissions += src.CommonCacheAdmissions + dst.CommonCacheAdmissionsSkipped += src.CommonCacheAdmissionsSkipped + dst.VoteCacheAdmissions += src.VoteCacheAdmissions + dst.VoteCacheAdmissionsSkipped += src.VoteCacheAdmissionsSkipped + dst.CachePublicationEpochRejects += src.CachePublicationEpochRejects + dst.DecodedAccountObjects += src.DecodedAccountObjects + dst.DecodedAccountBytes += src.DecodedAccountBytes + dst.PlaceholderObjects += src.PlaceholderObjects dst.WorkingSetLookup.AddTiming(time.Duration(src.WorkingSetLookupNanoseconds)) dst.InProgressLookup.AddTiming(time.Duration(src.InProgressNanoseconds)) dst.AppendVecPinWait.AddTiming(time.Duration(src.AppendVecPinWaitNanoseconds)) @@ -1295,6 +1293,18 @@ func reconstructFeeRateGovernor(s *state.MithrilState) *sealevel.FeeRateGovernor func configureBlock(block *b.Block, lastSlotCtx *sealevel.SlotCtx, epochSchedule *sealevel.SysvarEpochSchedule) error { + return configureBlockFromParent(block, lastSlotCtx, epochSchedule, true) +} + +// configureBlockFromParent derives the block's parent-dependent fields from +// the executed parent context. publishGlobal also publishes the block as the +// process-wide current slot (configureGlobalCtx); a speculative streaming +// shell passes false so the global view keeps describing the executed +// frontier until the complete block is configured. +func configureBlockFromParent(block *b.Block, + lastSlotCtx *sealevel.SlotCtx, + epochSchedule *sealevel.SysvarEpochSchedule, + publishGlobal bool) error { copy(block.ParentBankhash[:], lastSlotCtx.FinalBankhash) block.AcctsLtHash = lastSlotCtx.AcctsLtHash @@ -1310,7 +1320,9 @@ func configureBlock(block *b.Block, block.LastBlockhash = lastSlotCtx.Blockhash } - configureGlobalCtx(block) + if publishGlobal { + configureGlobalCtx(block) + } if global.ManageLeaderSchedule() { // epoch boundary. do not set leader @@ -1747,6 +1759,7 @@ func ReplayBlocks( var unwoundParentBankSysvars *sealevel.BankSysvars var partitionedEpochRewardsEnabled bool var partitionedRewardsInfo *rewards.PartitionedRewardDistributionInfo + var rewardsCompletion partitionedRewardsCompletion var featuresActivatedInFirstSlot []*accounts.Account var parentFeaturesActivatedInFirstSlot []*accounts.Account @@ -1928,8 +1941,8 @@ func ReplayBlocks( var highestExecutedSlot uint64 // highest slot ProcessBlock has executed; bounds the promotion-gate walk // While partitioned rewards distribute, promotion holds below the boundary // block so a crash-resume always re-runs it (the distribution bookkeeping is - // RAM-only and not reconstructible mid-window). Self-clears when the window - // completes (NumRewardPartitionsRemaining reaches 0). + // RAM-only and not reconstructible mid-window). Release requires a verified + // completion bank, committed atomically with the whole rewards window. var rewardsHoldBelowSlot uint64 // Alpenglow finality identities captured at observe/ingest time for the promotion // gate (the tracker's own state may be pruned by promotion time). Pruned as slots @@ -1989,9 +2002,9 @@ func ReplayBlocks( checkpointAfterCommit = consensusOpts.TransactionStatusCheckpointAfterCommit } if hookErr := unrootedTailState.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - // Snapshot runs here on the replay loop during fold-job construction; - // only its immutable bytes cross to the async worker. - Snapshot: transactionStatuses.SnapshotThrough, + // Pin the exact immutable view on replay. Sorting and encoding run + // on the existing fold worker, after releasing the live cache lock. + Capture: transactionStatuses.CaptureSnapshotThrough, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { return PrepareTransactionStatusCheckpoint(acctsDbPath, through, payload) }, @@ -2063,6 +2076,10 @@ func ReplayBlocks( mithrilState.LastRootedSlot = promotedThrough mithrilState.LastRootedBankhash = rootedCtx.Bankhash mithrilState.LastRootedContext = rootedCtx + if rewardsCompletion.retire(&partitionedRewardsInfo, promotedThrough) { + rewardsHoldBelowSlot = 0 + mlog.Log.Infof("epoch rewards bookkeeping retired through durable slot %d; later fork switches may unwind in memory", promotedThrough) + } if transactionStatuses.Root(promotedThrough) { mlog.Log.Infof("transaction status cache reconstructed complete %d-root coverage through durable slot %d", maxTransactionStatusRoots, promotedThrough) @@ -2170,13 +2187,9 @@ func ReplayBlocks( } promoteThrough := safePromoteTarget(lastRootedWatermark, verifierRequired, verifiedWM, replayDivergenceFloor) // Partitioned-rewards window: hold promotion below the boundary block - // until every partition distributes, so a crash-resume re-runs the - // boundary and rebuilds the RAM-only distribution bookkeeping. - if rewardsHoldBelowSlot > 0 && partitionedRewardsInfo != nil && partitionedRewardsInfo.NumRewardPartitionsRemaining > 0 { - if promoteThrough >= rewardsHoldBelowSlot { - promoteThrough = rewardsHoldBelowSlot - 1 - } - } + // until the completion bank verifies and is eligible to fold, so a + // failed distribution re-runs the boundary and rebuilds its bookkeeping. + promoteThrough = rewardsCompletion.limitPromotion(partitionedRewardsInfo, rewardsHoldBelowSlot, promoteThrough) if promoteThrough <= mithrilState.LastRootedSlot { // Operator signal: promotion is fully stalled (verifier lag, // divergence floor, or rewards hold) while finality has run at @@ -2207,6 +2220,8 @@ func ReplayBlocks( mlog.Log.FileOnlyf("alpenglow gate: checked=%d matched=%d no_finality=%d no_local_id=%d", gateStats.checked, gateStats.matched, gateStats.noFinality, gateStats.noLocalID) } + // The finality gate can stop before the verified completion bank. + promoteThrough = rewardsCompletion.limitPromotion(partitionedRewardsInfo, rewardsHoldBelowSlot, promoteThrough) if promoteThrough <= mithrilState.LastRootedSlot { return false } @@ -2220,6 +2235,18 @@ func ReplayBlocks( if res := promoter.drain(); res != nil { applyFoldOutcome(res) } + if rewardsHoldBelowSlot > 0 && partitionedRewardsInfo != nil && promoteThrough >= rewardsHoldBelowSlot { + job, jerr := unrootedTailState.buildRewardsCompletionFoldJob(rewardsCompletion.slot) + if jerr != nil { + mlog.Log.Errorf("rooted-durable: rewards completion fold: %v", jerr) + return false + } + if err := runFoldJob(unrootedTailState.committer, job); err != nil { + mlog.Log.Errorf("rooted-durable: rewards completion fold: %v", err) + return false + } + applyFoldOutcome(&foldResult{job: job}) + } promotedThrough, rootedCtx, perr := unrootedTailState.flush(promoteThrough) if perr != nil { mlog.Log.Errorf("rooted-durable: forced fold stopped at slot %d: %v", promotedThrough, perr) @@ -2233,7 +2260,13 @@ func ReplayBlocks( // when idle; completions are applied at the top of this function on a // later iteration. if !promoter.inFlight { - job, jerr := unrootedTailState.buildFoldJob(promoteThrough, false) + var job *foldJob + var jerr error + if rewardsHoldBelowSlot > 0 && partitionedRewardsInfo != nil && promoteThrough >= rewardsHoldBelowSlot { + job, jerr = unrootedTailState.buildRewardsCompletionFoldJob(rewardsCompletion.slot) + } else { + job, jerr = unrootedTailState.buildFoldJob(promoteThrough, false) + } if jerr != nil { mlog.Log.Errorf("rooted-durable: %v; watermark held back", jerr) return false @@ -2329,6 +2362,9 @@ func ReplayBlocks( opts.InitialAlpenglowBlockID = resumeState.ParentAlpenglowBlockID opts.HasInitialAlpenglowBlockID = true } + // Streaming execution needs the turbine batch feed; it is only ever + // eligible under Alpenglow with the unrooted tail (see streamingExecutor). + opts.TurbineStreamingExecution = StreamingExecutionCfg.Enabled && useTurbine && alpenglowMode && unrootedTailState != nil // Apply block fetching options if provided if blockFetchOpts != nil { @@ -2500,6 +2536,55 @@ func ReplayBlocks( } } + // Streaming execution: while the loop waits for the next complete block, + // the executor runs the entry batches of frontier+1 as turbine decodes + // them, against a speculative bank on lastSlotCtx. The closures read the + // loop's state at call time. streamInput is nil when the feed is off so + // the wait keeps its exact pre-streaming behaviour. + var streamer *streamingExecutor + var streamInput replayStreamer + // frontierMark is the timeline record of the last executed block (skips + // do not touch it: a child's parent is always a real block); the executor + // reads it to attribute a child's open delay to its parent's arrival, its + // parent's replay, or the loop itself. + var frontierMark streamingFrontierMark + if blockStream.StreamEvents() != nil { + streamer = newStreamingExecutor(streamingDeps{ + acctsDb: acctsDb, + feed: blockStream, + epochSchedule: epochSchedule, + txParallelism: txParallelism, + dbgOpts: dbgOpts, + persistedHashes: persistedHashes, + tail: unrootedTailState, + transactionStatuses: transactionStatuses, + alpenglowClock: alpenglowMode, + alpenglowMode: alpenglowMode, + unrootedTailUsed: unrootedTailState != nil, + lastSlotCtx: func() *sealevel.SlotCtx { return lastSlotCtx }, + frontier: func() uint64 { return replayFrontier }, + frontierMark: func() streamingFrontierMark { return frontierMark }, + currentFeatures: func() *features.Features { return replayCtx.CurrentFeatures }, + currentEpoch: func() uint64 { return currentEpoch }, + rewardsInFlight: func() bool { + return partitionedRewardsInfo != nil && partitionedRewardsInfo.NumRewardPartitionsRemaining > 0 + }, + switchPending: func() bool { + return switchSweeper.peek(alpenglowExecutedBlockIDs, mithrilState.LastRootedSlot, replayFrontier) != nil + }, + executedBlockID: func(slot uint64) (solana.Hash, bool) { + if id, ok := alpenglowExecutedBlockIDs[slot]; ok { + return id, true + } + return global.AlpenglowBlockID(slot) + }, + }) + streamInput = streamer + defer streamer.shutdown() + mlog.Log.Infof("streaming execution enabled: workers=%d min_group_batches=%d max_open=%s", + StreamingExecutionCfg.workers(txParallelism), StreamingExecutionCfg.MinGroupBatches, StreamingExecutionCfg.maxOpenAge()) + } + for { // The collector is per replay attempt. Discarded candidates, skipped // slots, and typed-recovery exits must never leak timings into the next @@ -2518,6 +2603,7 @@ func ReplayBlocks( ingressTimings *b.TurbineIngressTimings waitTime time.Duration neededAt time.Time // when replay asked the source for this slot + admittedAt time.Time // when the source handed replay this input ) { @@ -2551,14 +2637,20 @@ func ReplayBlocks( } neededAt = time.Now() - block, parentSwitch, certifiedSwitch = waitForAlpenglowReplayInput(ctx, - blockStream.NextBlockOrAlpenglowEvent, sweepWhileWaiting, decisionChanges, alpenglowSwitchPollInterval) + if frontierMark.waitEnteredAt.IsZero() { + // First wait after the last executed block: what precedes it is + // that block's post-replay tail (promotion, RPC, stats). + frontierMark.waitEnteredAt = neededAt + } + block, parentSwitch, certifiedSwitch = waitForReplayInput(ctx, + blockStream.NextReplayInput, sweepWhileWaiting, decisionChanges, alpenglowSwitchPollInterval, streamInput) if ingress, ok := block.CompleteTurbineReplayAdmission(time.Now()); ok { _ = statsd.Duration(statsd.TurbineReplayAdmission, ingress.ReplayAdmission, nil) ingressTimings = &ingress } - waitTime = time.Since(neededAt) + admittedAt = time.Now() + waitTime = admittedAt.Sub(neededAt) if stallDone != nil { close(stallDone) @@ -2568,6 +2660,11 @@ func ReplayBlocks( result.WasCancelled = true break } + // Any fork switch invalidates a speculative bank above the frontier: + // its parent chain is about to be unwound or re-served. + if certifiedSwitch != nil || parentSwitch != nil { + streamer.discard("fork_switch") + } if certifiedSwitch != nil { if handleAlpenglowSwitch(certifiedSwitch, func() bool { blockStream.RewindForAlpenglowSwitch(certifiedSwitch.Slot, certifiedSwitch.Certified) @@ -2653,10 +2750,16 @@ func ReplayBlocks( continue } + // A speculative stream may span unresolved slots on the last bank. + // An actual intervening block invalidates that assumption before any + // validation or bank work can observe speculative global state. + streamer.beforeBlock(block) + // An in-flight source send can race the first quarantine drain. Exact // emitted suffix IDs are hard-tombstoned before that send, so discard // any leaked descendant before it reaches consensus observation. if blockStream.IsObjectivelyInvalidAlpenglowBlock(block) { + streamer.discardSlot(block.Slot, "quarantined") mlog.Log.Warnf("replay: discarding quarantined Alpenglow block %s at slot %d before consensus observation", solana.Hash(block.AlpenglowBlockID), block.Slot) continue @@ -2666,6 +2769,7 @@ func ReplayBlocks( if validationErr := validatePreConsensusTransactionStatuses( transactionStatuses, block, currentExecutedAnchorSlot(), ); validationErr != nil { + streamer.discardSlot(block.Slot, "status_validation") if !IsAlreadyProcessedTransactionError(validationErr) { result.Error = fmt.Errorf("pre-consensus block validation failed at slot %d: %w", block.Slot, validationErr) mlog.Log.Errorf("%v", result.Error) @@ -2687,6 +2791,7 @@ func ReplayBlocks( // or any bank changes, while the selected parent is still untouched. if alpenglowMode && !block.IsSkipped { if validationErr := validatePreConsensusRewardCertificates(block, epochSchedule, block.AlpenglowShredVersion); validationErr != nil { + streamer.discardSlot(block.Slot, "reward_certificates") if !IsInvalidRewardCertificateError(validationErr) { result.Error = fmt.Errorf("pre-consensus reward validation failed at slot %d: %w", block.Slot, validationErr) mlog.Log.Errorf("%v", result.Error) @@ -2735,6 +2840,7 @@ func ReplayBlocks( // selected block in either case. if unrootedTailState != nil { if sw := switchSweeper.sweep(alpenglowExecutedBlockIDs, mithrilState.LastRootedSlot, replayFrontier); sw != nil { + streamer.discard("fork_switch") if handleAlpenglowSwitch(sw, func() bool { blockStream.RewindForAlpenglowSwitch(sw.Slot, sw.Certified) return true @@ -2790,6 +2896,7 @@ func ReplayBlocks( // Handle skipped slots - log and continue without execution if block.IsSkipped { + streamer.discardSlot(block.Slot, "skipped") // Zero is the explicit locally consumed outcome for a skip. Parent-ID // gap inference is provisional; recording it lets a later certificate // or discovered ancestry require a source rewind and, when necessary, @@ -2836,6 +2943,11 @@ func ReplayBlocks( record.TransactionParse.AddTiming(ingressTimings.TransactionParse) record.TransactionSigverify.AddTiming(ingressTimings.TransactionSigverify) record.ReplayAdmission.AddTiming(ingressTimings.ReplayAdmission) + record.EarlyTransactionParse.AddTiming(ingressTimings.EarlyTransactionParse) + record.EarlyTransactionSigverify.AddTiming(ingressTimings.EarlyTransactionSigverify) + record.EarlyPreparationWait.AddTiming(ingressTimings.EarlyPreparationWait) + record.EarlyVerifiedTransactions = ingressTimings.EarlyVerifiedTransactions + record.FullToReady.AddTiming(ingressTimings.FullToReady) } start := time.Now() @@ -2914,6 +3026,7 @@ func ReplayBlocks( boundaryParentCtx = epochBoundaryParentCtx(acctsDb, block, currentEpoch, replayCtx.CurrentFeatures) } partitionedRewardsInfo = handleEpochTransition(acctsDb, partitionedEpochRewardsEnabled, boundaryParentCtx, replayCtx, epochSchedule, replayCtx.CurrentFeatures, block, currentEpoch, rpcc, dbgOpts) + rewardsCompletion = partitionedRewardsCompletion{} currentEpoch = block.Epoch justCrossedEpochBoundary = true // While partitioned rewards are distributing, hold durable promotion @@ -3011,9 +3124,27 @@ func ReplayBlocks( parentBankSysvars = lastSlotCtx.BankSysvars() } if block.FromLocalProduction { + streamer.discardSlot(block.Slot, "local_production") lastSlotCtx, err = adoptLocalLeaderBlock(block, unrootedTailState, transactionStatuses, persistedHashes) } else { - lastSlotCtx, err = ProcessBlock(acctsDb, block, epochSchedule, txParallelism, dbgOpts, persistedHashes, unrootedTailState, transactionStatuses, alpenglowClock, parentBankSysvars) + // A stream open on this slot finishes the block on its speculative + // bank when the block proves to be what it executed; otherwise the + // stream is discarded and the block executes whole, exactly as + // without streaming. + streamed := false + if streamer.matches(block.Slot) { + var streamedCtx *sealevel.SlotCtx + streamedCtx, streamed, err = streamer.finalize(block, parentBankSysvars) + if streamed { + lastSlotCtx = streamedCtx + } + } else { + streamer.discard("other_block") + } + if !streamed { + streamer.noteWholeBlock(block) + lastSlotCtx, err = ProcessBlock(acctsDb, block, epochSchedule, txParallelism, dbgOpts, persistedHashes, unrootedTailState, transactionStatuses, alpenglowClock, parentBankSysvars) + } } processBlockEnd := time.Now() metrics.GlobalBlockReplay.ProcessBlock.AddTiming(processBlockEnd.Sub(processBlockStart)) @@ -3026,6 +3157,7 @@ func ReplayBlocks( } // The successful child now owns its derived snapshot. Any later bank uses // lastSlotCtx; the one-shot retained unwind bridge is no longer needed. + rewardsCompletion.observeBank(partitionedRewardsInfo, lastSlotCtx.BankSysvars()) unwoundParentBankSysvars = nil postProcessBlockStart := processBlockEnd statusViewStart := time.Now() @@ -3082,6 +3214,11 @@ func ReplayBlocks( break } } + recordFullToReplayed(block) + // The same instant FullToReplayed ends at: from here to the next wait + // entry is this block's post-replay tail, which a child's open timeline + // reports as OpenWaitPostReplay. + frontierMark = streamingFrontierMark{slot: block.Slot, fullNanos: block.ShredFullNanos, admittedAt: admittedAt, replayedAt: time.Now()} if rpcServer != nil { rpcServer.SetSlotCtx(lastSlotCtx) @@ -4154,71 +4291,31 @@ func ProcessBlock( return nil, fmt.Errorf("validate transaction messages for slot %d: %w", block.Slot, err) } statusValidationStart := time.Now() - statusValidationErr := transactionStatuses.validateBlockWithPlan(block, executionPlan) + statusValidation, statusValidationErr := transactionStatuses.validateBlockForPublication(block, executionPlan) metrics.GlobalBlockReplay.TransactionStatusValidation.AddTimingSince(statusValidationStart) if statusValidationErr != nil { return nil, fmt.Errorf("validate transaction statuses for slot %d: %w", block.Slot, statusValidationErr) } - ctx, task := trace.NewTask(context.Background(), "ProcessBlock") - defer task.End() - trace.Log(ctx, "slot", fmt.Sprintf("%d", block.Slot)) - trace.Log(ctx, "txCount", fmt.Sprintf("%d", len(block.Transactions))) - - var replayStage atomic.Value - var replayStageSince atomic.Int64 - setReplayStage := func(stage string) { - replayStage.Store(stage) - replayStageSince.Store(time.Now().UnixNano()) - } - setReplayStage("prepare_dependency_planner") - - replayWatchdogDone := make(chan struct{}) - go func() { - ticker := time.NewTicker(5 * time.Second) - defer ticker.Stop() - - var lastLoggedStage string - var lastLoggedSince int64 - for { - select { - case <-replayWatchdogDone: - return - case <-ticker.C: - stageVal := replayStage.Load() - stage, ok := stageVal.(string) - if !ok || stage == "" { - continue - } - sinceUnix := replayStageSince.Load() - if sinceUnix == 0 { - continue - } - if stage == lastLoggedStage && sinceUnix == lastLoggedSince { - continue - } - stageDuration := time.Since(time.Unix(0, sinceUnix)) - if stageDuration < 10*time.Second { - continue - } - mlog.Log.Warnf("REPLAY WATCHDOG: slot %d stuck in stage %s for %s | txs=%d | lightbringer=%t", - block.Slot, stage, stageDuration.Round(time.Second), len(block.Transactions), block.FromLiveStream) - lastLoggedStage = stage - lastLoggedSince = sinceUnix - } + statusPreparation := transactionStatuses.startStatusPreparation(executionPlan) + defer func() { + // Join before returning so a rejected bank cannot leave work behind or + // charge its preparation time to the next block's metrics record. + statusPreparation.wait() + if statusPreparation != nil { + metrics.GlobalBlockReplay.TransactionStatusPreparation.AddTiming(statusPreparation.duration) } }() - defer close(replayWatchdogDone) + + // The resumable execution state carries the trace task, stage watchdog and + // the SlotCtx; streaming execution drives the same object group by group. + exec := newBlockExecution(acctsDb, block, epochSchedule, txParallelism, dbgOpts, persistedHashes, tail, transactionStatuses, alpenglowClock, parentBankSysvars) + defer exec.close() + exec.setReplayStage("prepare_dependency_planner") if SerializedParameterArena != nil { SerializedParameterArena.Reset() } - var sigverifyWg sync.WaitGroup - defer func() { - sigverifyJoinStart := time.Now() - sigverifyWg.Wait() - metrics.GlobalBlockReplay.SignatureVerificationJoin.AddTimingSince(sigverifyJoinStart) - }() plannerPreparationStart := time.Now() var planner *preparedDependencyPlanner if txParallelism > 0 { @@ -4231,15 +4328,9 @@ func ProcessBlock( metrics.GlobalBlockReplay.DependencyPlannerPreparation.AddTimingSince(plannerPreparationStart) start := time.Now() - setReplayStage("load_accounts") - loadAcctsRegion := trace.StartRegion(ctx, "LoadBlockAccounts") - // In rooted-durable mode, block accounts/sysvars load through the unrooted - // tail (overlay→durable) so execution sees confirmed-but-unrooted state. - var blockSrc blockAccountSource = acctsDb - if tail != nil { - blockSrc = tail - } - accts, parentAccts, accountMapCapacity, bankSysvars, err := loadBlockAccountsAndUpdateSysvars(blockSrc, block, epochSchedule, alpenglowClock, parentBankSysvars, planner) + exec.setReplayStage("load_accounts") + loadAcctsRegion := trace.StartRegion(exec.ctx, "LoadBlockAccounts") + accts, parentAccts, accountMapCapacity, bankSysvars, err := loadBlockAccountsAndUpdateSysvars(exec.blockSrc, block, epochSchedule, alpenglowClock, parentBankSysvars, planner) loadAcctsRegion.End() if err != nil { panic(fmt.Sprintf("unable to load slot accounts and update sysvars: %s", err)) @@ -4250,181 +4341,53 @@ func ProcessBlock( metrics.GlobalBlockReplay.LoadBlockAccounts.AddTimingSince(start) slotCtxSetupStart := time.Now() - slotCtx := newSlotCtx(block, accts, parentAccts, acctsDb, tail, accountMapCapacity) - if err := slotCtx.PublishBankSysvars(bankSysvars); err != nil { - return nil, fmt.Errorf("publish bank sysvars at slot %d: %w", block.Slot, err) + if err := exec.installSlotCtx(accts, parentAccts, accountMapCapacity, bankSysvars); err != nil { + return nil, err } - bankEpochScheduleValue, ok := bankSysvars.EpochSchedule() - if !ok { - return nil, fmt.Errorf("bank-local EpochSchedule sysvar unavailable at slot %d", block.Slot) - } - bankEpochSchedule := &bankEpochScheduleValue + slotCtx := exec.slotCtx if requireAlpenglowBlockFooter(block, slotCtx, alpenglowClock) { if err := validateAlpenglowFooterNanosecondClock(slotCtx, block); err != nil { return nil, err } } - slotCtx.TraceCtx = ctx slotCtx.NumSignatures = executionPlan.processedSignatures metrics.GlobalBlockReplay.SlotCtxSetup.AddTimingSince(slotCtxSetupStart) var txFeeAccumulator fees.TxFeeInfoAccumulator var totalComputeUnitsConsumed uint64 start = time.Now() - setReplayStage("tx_loop") - txLoopRegion := trace.StartRegion(ctx, "TxLoop") + exec.setReplayStage("tx_loop") + txLoopRegion := trace.StartRegion(exec.ctx, "TxLoop") shouldVerifySignatures := !block.TransactionSignaturesVerified() if txParallelism > 0 { - txFeeAccumulator, totalComputeUnitsConsumed = parallelTxLoop(slotCtx, &sigverifyWg, planner, block, executionPlan, txParallelism, dbgOpts, shouldVerifySignatures) + txFeeAccumulator, totalComputeUnitsConsumed = parallelTxLoop(slotCtx, &exec.sigverifyWg, planner, block, executionPlan, txParallelism, dbgOpts, shouldVerifySignatures) } else { - txFeeAccumulator, totalComputeUnitsConsumed = sequentialTxLoop(slotCtx, &sigverifyWg, block, executionPlan, dbgOpts, shouldVerifySignatures) + txFeeAccumulator, totalComputeUnitsConsumed = sequentialTxLoop(slotCtx, &exec.sigverifyWg, block, executionPlan, dbgOpts, shouldVerifySignatures) } slotCtx.TotalComputeUnitsConsumed = totalComputeUnitsConsumed txLoopRegion.End() metrics.GlobalBlockReplay.TxLoop.AddTimingSince(start) - start = time.Now() - setReplayStage("distribute_fees") - - // distribute tx fees to the slot leader - // skip leader handling if there are zero transactions in this block - if !global.ManageLeaderSchedule() && block.BlockReward != nil && len(block.Transactions) > 0 { - slotCtx.LamportsBurnt = fees.DistributeTxFeesToSlotLeader(acctsDb, slotCtx, block.BlockReward.Leader, &txFeeAccumulator) - slotCtx.RecordModifiedAcct(block.BlockReward.Leader) - } else if global.ManageLeaderSchedule() && len(block.Transactions) > 0 { - slotCtx.LamportsBurnt = fees.DistributeTxFeesToSlotLeader(acctsDb, slotCtx, block.Leader, &txFeeAccumulator) - slotCtx.RecordModifiedAcct(block.Leader) - } - metrics.GlobalBlockReplay.Reward.AddTimingSince(start) - - start = time.Now() - setReplayStage("collect_rent") - bankRent, ok := slotCtx.BankSysvars().Rent() - if !ok { - return nil, fmt.Errorf("bank-local Rent sysvar unavailable at slot %d", block.Slot) - } - rentAccts := rent.CollectRentEagerly(slotCtx, &bankRent, bankEpochSchedule) - metrics.GlobalBlockReplay.Rent.AddTimingSince(start) - - start = time.Now() - setReplayStage("run_incinerator") - runIncinerator(slotCtx) - metrics.GlobalBlockReplay.RunIncinerator.AddTimingSince(start) - - // Alpenglow banks set the Clock timestamp from the block footer after execution. - if alpenglowClock { - footerClockStart := time.Now() - if err := applyAlpenglowFooterClock(slotCtx, block, bankEpochSchedule); err != nil { - metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) - return nil, fmt.Errorf("apply alpenglow footer clock at slot %d: %w", block.Slot, err) - } - if err := updateAlpenglowNanosecondClockAccount(slotCtx, block); err != nil { - metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) - return nil, err - } - metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) - voteRewardsStart := time.Now() - voteRewardsErr := ApplyAlpenglowVoteRewards(slotCtx, block, bankEpochSchedule, block.SkipRewardCert, block.NotarRewardCert, block.BlockFinalCert, block.AlpenglowShredVersion) - metrics.GlobalBlockReplay.AlpenglowVoteRewards.AddTimingSince(voteRewardsStart) - if voteRewardsErr != nil { - return nil, voteRewardsErr - } - } - if err := finalizeBankSysvars(slotCtx); err != nil { - return nil, fmt.Errorf("finalize bank sysvars at slot %d: %w", block.Slot, err) - } - - setReplayStage("compile_accounts") - start = time.Now() - writableAccts, modifiedAccts := compileWritableAndModifiedAccts(slotCtx, block, rentAccts) - metrics.GlobalBlockReplay.CompileWritableAndModifiedAccts.AddTimingSince(start) - start = time.Now() - ensureParentsErr := ensureParentAccountsForModified(slotCtx, modifiedAccts) - metrics.GlobalBlockReplay.EnsureParentAccountsForModified.AddTimingSince(start) - if ensureParentsErr != nil { - return nil, ensureParentsErr - } - - start = time.Now() - setReplayStage("bankhash") - slotCtx.FinalBankhash = bankhash.CalculateBankHash(slotCtx, writableAccts, modifiedAccts, block.ParentBankhash, slotCtx.NumSignatures, block.Blockhash) - metrics.GlobalBlockReplay.BankHash.AddTimingSince(start) - if alpenglowClock { - footerVerificationStart := time.Now() - footerVerificationErr := verifyAlpenglowBlockFooter(slotCtx, block, alpenglowClock) - metrics.GlobalBlockReplay.AlpenglowFooterVerification.AddTimingSince(footerVerificationStart) - if footerVerificationErr != nil { - writeFooterBankhashMismatchArtifact(footerVerificationErr, block, slotCtx, writableAccts, modifiedAccts) - return nil, footerVerificationErr - } - } - - // Bankhash consensus enforcement is handled in the replay loop (not here) - // because forkchoice is fed after ProcessBlock returns — checking here would - // never see votes from recently submitted blocks and could deadlock. - - // Enter critical commit window - panics here may leave AccountsDB inconsistent - commitSlot.Store(slotCtx.Slot) - commitInProgress.Store(true) - blockUpdateStart := time.Now() - setReplayStage("store_accounts") - persistedSlot := slotCtx.Slot - persistedBankhash := append([]byte(nil), slotCtx.FinalBankhash...) - persistedBlockSlot := block.Slot - stakeIndexDir := filepath.Join(acctsDb.AcctsDir, "..") - afterStoreAccounts := func() { - if tail != nil { - // Rooted-durable: accounts + bankhash are buffered in the overlay and - // become durable only on promotion; nothing written here (rooted-only). - } else { - if berr := acctsDb.StoreBankHashForSlot(persistedSlot, persistedBankhash); berr != nil { - mlog.Log.Infof("unable to store bankhash for slot %d", persistedSlot) - } - } - if tail == nil { - // Legacy/verify modes (no fork ambiguity): flush per block as before. - // Rooted-durable replay flushes at FOLD time instead — entries stay - // slot-scoped in RAM so a fork unwind can drop them, and scans merge - // the pending set (StreamStakeAccounts) for completeness meanwhile. - flushed, err := global.FlushPendingStakePubkeys(stakeIndexDir) - if err != nil { - mlog.Log.Errorf("failed to flush stake pubkey index: %v", err) - } else if flushed > 0 { - mlog.Log.Debugf("flushed %d new stake pubkeys to index", flushed) - } - } - - persistedHashes.Set(persistedBlockSlot, persistedBankhash) - - // Exit critical commit window - AccountsDB is now consistent - commitInProgress.Store(false) - commitSlot.Store(0) - } + exec.txFeeAccumulator = txFeeAccumulator + exec.totalCU = totalComputeUnitsConsumed + exec.executionPlan = executionPlan + exec.statusPreparation = statusPreparation + exec.statusValidation = statusValidation + return exec.finalize() +} - if tail != nil { - // Rooted-durable: buffer this slot's writes + bankhash in the RAM overlay - // (always, even when empty, so the bankhash is recorded); no durable write. - tail.Add(slotCtx.Slot, modifiedAccts, persistedBankhash) - afterStoreAccounts() - } else if len(modifiedAccts) > 0 { - err = acctsDb.StoreAccounts(modifiedAccts, slotCtx.Slot, afterStoreAccounts) - } - // In rooted-durable mode the callback above is synchronous, so this includes - // the complete critical-path overlay publication. Legacy StoreAccounts only - // enqueues here; its asynchronous disk work deliberately belongs to no slot's - // replay wall time and must never update a later slot's metrics record. - metrics.GlobalBlockReplay.BlockUpdateAccounts.AddTimingSince(blockUpdateStart) - if err != nil { - return slotCtx, err +// recordFullToReplayed measures the vote-path latency replay controls for a +// turbine block: from the assembler's full-assembly instant (the last shred, +// carried as ShredFullNanos) to the replay result reaching consensus. Blocks +// that did not arrive as shreds carry no full instant and record nothing. +func recordFullToReplayed(block *b.Block) { + if block == nil || block.ShredFullNanos <= 0 { + return } - statusCommitStart := time.Now() - statusErr := transactionStatuses.commitBlockWithPlan(block, executionPlan) - metrics.GlobalBlockReplay.TransactionStatusCommit.AddTimingSince(statusCommitStart) - if statusErr != nil { - return nil, fmt.Errorf("commit transaction statuses for slot %d after bank state commit: %w", block.Slot, statusErr) + fullToReplayed := time.Since(time.Unix(0, block.ShredFullNanos)) + if fullToReplayed <= 0 { + return } - - global.IncrTransactionCount(executionPlan.processedTxCount) - setReplayStage("done") - return slotCtx, err + metrics.GlobalBlockReplay.FullToReplayed.AddTiming(fullToReplayed) + _ = statsd.Duration(statsd.ReplayFullToReplayed, fullToReplayed, nil) } diff --git a/pkg/replay/block_execution.go b/pkg/replay/block_execution.go new file mode 100644 index 000000000..5e5e5518e --- /dev/null +++ b/pkg/replay/block_execution.go @@ -0,0 +1,699 @@ +package replay + +import ( + "context" + "errors" + "fmt" + "path/filepath" + "runtime/trace" + "sync" + "sync/atomic" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/Overclock-Validator/mithril/pkg/arena" + "github.com/Overclock-Validator/mithril/pkg/bankhash" + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/fees" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/Overclock-Validator/mithril/pkg/rent" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" +) + +// blockExecution is the resumable state of one bank's execution. ProcessBlock +// drives it in one pass (plan, load, execute every transaction, finalize). +// Streaming execution opens it before the block is complete, feeds +// transaction groups as their shreds arrive, and finalizes against the +// complete block. Both paths share the opening (bank sysvars, SlotCtx) and the +// tail (fees, rent, footer, bank hash, commit), so a bank produced either way +// runs the same end-of-block code over the same SlotCtx. +type blockExecution struct { + acctsDb *accountsdb.AccountsDb + block *b.Block + epochSchedule *sealevel.SysvarEpochSchedule + txParallelism int + dbgOpts *DebugOptions + persistedHashes *persistedTracker + tail unrootedState + transactionStatuses *TransactionStatusCache + alpenglowClock bool + parentBankSysvars *sealevel.BankSysvars + + blockSrc blockAccountSource + slotCtx *sealevel.SlotCtx + parentAccts accounts.MemAccounts + accts accounts.Accounts + bankSysvars *sealevel.BankSysvars + bankEpochSchedule *sealevel.SysvarEpochSchedule + + ctx context.Context + task *trace.Task + setReplayStage func(string) + watchdogDone chan struct{} + sigverifyWg sync.WaitGroup + closed bool + + // Whole-block inputs to the tail, set by ProcessBlock. + executionPlan blockTransactionExecutionPlan + statusPreparation *transactionStatusPreparation + statusValidation transactionStatusValidation + + // Incremental transaction bookkeeping in block order, maintained by + // executeTransactionGroup. ProcessBlock does not use it. + transactions []*solana.Transaction + identities []txstatus.TransactionMessageIdentity + execute []bool + seenMessages map[[32]byte]int + processedTxCount uint64 + processedSignatures uint64 + groups int + + txFeeAccumulator fees.TxFeeInfoAccumulator + totalCU uint64 + + // Retained across global metric resets while a speculative stream waits. + accountLoader metrics.AccountLoader +} + +// newBlockExecution installs the per-bank trace task, the stage watchdog and +// the account source; it does not touch bank state. +func newBlockExecution( + acctsDb *accountsdb.AccountsDb, + block *b.Block, + epochSchedule *sealevel.SysvarEpochSchedule, + txParallelism int, + dbgOpts *DebugOptions, + persistedHashes *persistedTracker, + tail unrootedState, + transactionStatuses *TransactionStatusCache, + alpenglowClock bool, + parentBankSysvars *sealevel.BankSysvars, +) *blockExecution { + exec := &blockExecution{ + acctsDb: acctsDb, + block: block, + epochSchedule: epochSchedule, + txParallelism: txParallelism, + dbgOpts: dbgOpts, + persistedHashes: persistedHashes, + tail: tail, + transactionStatuses: transactionStatuses, + alpenglowClock: alpenglowClock, + parentBankSysvars: parentBankSysvars, + seenMessages: make(map[[32]byte]int), + } + // In rooted-durable mode, block accounts/sysvars load through the unrooted + // tail (overlay→durable) so execution sees confirmed-but-unrooted state. + exec.blockSrc = acctsDb + if tail != nil { + exec.blockSrc = tail + } + + ctx, task := trace.NewTask(context.Background(), "ProcessBlock") + exec.ctx, exec.task = ctx, task + trace.Log(ctx, "slot", fmt.Sprintf("%d", block.Slot)) + trace.Log(ctx, "txCount", fmt.Sprintf("%d", len(block.Transactions))) + + var replayStage atomic.Value + var replayStageSince atomic.Int64 + exec.setReplayStage = func(stage string) { + replayStage.Store(stage) + replayStageSince.Store(time.Now().UnixNano()) + } + + exec.watchdogDone = make(chan struct{}) + go func() { + ticker := time.NewTicker(5 * time.Second) + defer ticker.Stop() + + var lastLoggedStage string + var lastLoggedSince int64 + for { + select { + case <-exec.watchdogDone: + return + case <-ticker.C: + stageVal := replayStage.Load() + stage, ok := stageVal.(string) + if !ok || stage == "" { + continue + } + sinceUnix := replayStageSince.Load() + if sinceUnix == 0 { + continue + } + if stage == lastLoggedStage && sinceUnix == lastLoggedSince { + continue + } + stageDuration := time.Since(time.Unix(0, sinceUnix)) + if stageDuration < 10*time.Second { + continue + } + mlog.Log.Warnf("REPLAY WATCHDOG: slot %d stuck in stage %s for %s | txs=%d | lightbringer=%t", + block.Slot, stage, stageDuration.Round(time.Second), len(block.Transactions), block.FromLiveStream) + lastLoggedStage = stage + lastLoggedSince = sinceUnix + } + } + }() + return exec +} + +// close joins outstanding signature verification, stops the watchdog and ends +// the trace task, in the order ProcessBlock's deferred cleanup always used. It +// is idempotent so a discarded stream and a finalized bank both call it. +func (exec *blockExecution) close() { + if exec == nil || exec.closed { + return + } + exec.closed = true + sigverifyJoinStart := time.Now() + exec.sigverifyWg.Wait() + metrics.GlobalBlockReplay.SignatureVerificationJoin.AddTimingSince(sigverifyJoinStart) + if exec.watchdogDone != nil { + close(exec.watchdogDone) + } + if exec.task != nil { + exec.task.End() + } +} + +// installSlotCtx publishes the loaded parent snapshot and derived bank sysvars +// as this bank's SlotCtx. The overlay/parent pair comes from +// loadBlockAccountsAndUpdateSysvars; streaming grows the parent snapshot +// afterwards, group by group, through loadTransactionAccounts. +func (exec *blockExecution) installSlotCtx(accts accounts.Accounts, parentAccts accounts.Accounts, accountMapCapacity int, bankSysvars *sealevel.BankSysvars) error { + block := exec.block + slotCtx := newSlotCtx(block, accts, parentAccts, exec.acctsDb, exec.tail, accountMapCapacity) + if err := slotCtx.PublishBankSysvars(bankSysvars); err != nil { + return fmt.Errorf("publish bank sysvars at slot %d: %w", block.Slot, err) + } + bankEpochScheduleValue, ok := bankSysvars.EpochSchedule() + if !ok { + return fmt.Errorf("bank-local EpochSchedule sysvar unavailable at slot %d", block.Slot) + } + slotCtx.TraceCtx = exec.ctx + exec.slotCtx = slotCtx + exec.accts = accts + if mem, ok := parentAccts.(accounts.MemAccounts); ok { + exec.parentAccts = mem + } + exec.bankSysvars = bankSysvars + exec.bankEpochSchedule = &bankEpochScheduleValue + return nil +} + +// open performs the bank-start work that needs only the parent state: the +// parent sysvar pin and this bank's Clock/SlotHashes derivation, the parent +// snapshot for whatever transactions the block currently carries (none for a +// streaming shell), the overlay, and the SlotCtx. It is ProcessBlock's opening +// without the whole-block planner, so a streaming caller can start executing +// groups before any transaction of the block is known. +func (exec *blockExecution) open() error { + defer exec.captureAccountLoader()() + block := exec.block + exec.setReplayStage("prepare_dependency_planner") + if SerializedParameterArena != nil { + SerializedParameterArena.Reset() + } + + start := time.Now() + exec.setReplayStage("load_accounts") + loadAcctsRegion := trace.StartRegion(exec.ctx, "LoadBlockAccounts") + accts, parentAccts, accountMapCapacity, bankSysvars, err := loadBlockAccountsAndUpdateSysvars(exec.blockSrc, block, exec.epochSchedule, exec.alpenglowClock, exec.parentBankSysvars, nil) + loadAcctsRegion.End() + if err != nil { + return fmt.Errorf("load slot accounts and update sysvars at slot %d: %w", block.Slot, err) + } + if err := bankSysvars.ValidateForExecution(); err != nil { + return fmt.Errorf("invalid bank sysvar snapshot at slot %d: %w", block.Slot, err) + } + metrics.GlobalBlockReplay.LoadBlockAccounts.AddTimingSince(start) + + slotCtxSetupStart := time.Now() + if err := exec.installSlotCtx(accts, parentAccts, accountMapCapacity, bankSysvars); err != nil { + return err + } + metrics.GlobalBlockReplay.SlotCtxSetup.AddTimingSince(slotCtxSetupStart) + return nil +} + +// errBlockExecutionClosed reports a group offered after close or finalize. +var errBlockExecutionClosed = errors.New("block execution is closed") + +// groupIdentitiesFor returns prepared identities for a transaction group, +// hashing the messages when the caller has none from signature verification. +func groupIdentitiesFor(txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities) (*b.PreparedTransactionMessageIdentities, error) { + if identities != nil { + if identities.Len() != len(txs) { + return nil, fmt.Errorf("group identities cover %d transactions, group has %d", identities.Len(), len(txs)) + } + return identities, nil + } + view := &b.Block{Transactions: txs} + return view.PrepareTransactionMessageIdentities() +} + +// executeTransactionGroup executes the next transactions of the block, in +// block order, against the open SlotCtx. Every check ProcessBlock applies to a +// whole block is applied incrementally: message versions against the bank's +// features, duplicate messages across every group so far (a duplicate makes +// the whole block invalid, exactly as planBlockTransactionExecution reports +// it), ancestor status-cache validation, address-table resolution, account +// loading into the same parent snapshot, and a dependency plan over the group +// executed by up to txParallelism workers. Groups run strictly one after +// another, so cross-group ordering is the sequential block order. +// +// identities may carry the verifier's message identities for exactly these +// transactions; nil hashes them here. shouldVerifySignatures is passed to +// ProcessTransaction unchanged. +func (exec *blockExecution) executeTransactionGroup(txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerifySignatures bool) error { + if exec == nil || exec.slotCtx == nil { + return errors.New("block execution is not open") + } + if exec.closed { + return errBlockExecutionClosed + } + if len(txs) == 0 { + return nil + } + block := exec.block + slot := block.Slot + base := len(exec.transactions) + + view := &b.Block{Slot: slot, Transactions: txs, Features: block.Features} + if err := validateBlockTransactionVersions(view); err != nil { + return fmt.Errorf("validate transaction versions for slot %d: %w", slot, err) + } + + prepared, err := groupIdentitiesFor(txs, identities) + if err != nil { + return fmt.Errorf("validate transaction messages for slot %d: %w", slot, err) + } + execute := make([]bool, len(txs)) + var duplicates *DuplicateTransactionMessagesError + for idx, tx := range txs { + if tx == nil { + return fmt.Errorf("validate transaction messages for slot %d: transaction %d is nil", slot, base+idx) + } + identity := prepared.Identity(idx) + if firstIndex, duplicate := exec.seenMessages[identity.MessageHash]; duplicate { + if duplicates == nil { + duplicates = &DuplicateTransactionMessagesError{Slot: slot} + } + duplicates.DuplicateCount++ + if len(duplicates.Occurrences) < maxDuplicateTransactionOccurrences { + duplicates.Occurrences = append(duplicates.Occurrences, DuplicateTransactionOccurrence{ + Index: base + idx, FirstIndex: firstIndex, + }) + } + continue + } + exec.seenMessages[identity.MessageHash] = base + idx + execute[idx] = true + } + if duplicates != nil { + return fmt.Errorf("validate transaction messages for slot %d: %w", slot, duplicates) + } + if exec.transactionStatuses != nil { + if err := exec.transactionStatuses.validateTransactionsAgainstAncestors(slot, prepared); err != nil { + return fmt.Errorf("validate transaction statuses for slot %d: %w", slot, err) + } + } + + // Record the group before executing so a failure after this point still + // leaves the block-order view consistent for finalize's prefix proof. + for idx, tx := range txs { + exec.transactions = append(exec.transactions, tx) + exec.identities = append(exec.identities, prepared.Identity(idx)) + exec.execute = append(exec.execute, execute[idx]) + if execute[idx] { + exec.processedTxCount++ + exec.processedSignatures += uint64(tx.Message.Header.NumRequiredSignatures) + } + } + exec.slotCtx.NumSignatures = exec.processedSignatures + + exec.setReplayStage("load_accounts") + if err := exec.loadTransactionAccounts(view); err != nil { + return err + } + + exec.setReplayStage("tx_loop") + start := time.Now() + txLoopRegion := trace.StartRegion(exec.ctx, "TxLoop") + feeInfos, computeUnits, err := exec.runTransactionGroup(txs, execute, shouldVerifySignatures) + txLoopRegion.End() + metrics.GlobalBlockReplay.TxLoop.AddTimingSince(start) + if err != nil { + return err + } + for idx, txFeeInfo := range feeInfos { + if !execute[idx] { + continue + } + exec.totalCU += computeUnits[idx] + exec.txFeeAccumulator.Add(txFeeInfo) + } + exec.slotCtx.TotalComputeUnitsConsumed = exec.totalCU + exec.groups++ + metrics.GlobalBlockReplay.StreamingExecution.Groups++ + metrics.GlobalBlockReplay.StreamingExecution.Transactions += uint64(len(txs)) + return nil +} + +// captureAccountLoader isolates this stream's loader work from the global +// record, which may belong to another replayed slot or be reset while waiting. +// Only the replay goroutine may enter this scope; loader workers are joined +// before it exits. Discarded streams never publish their retained totals. +func (exec *blockExecution) captureAccountLoader() func() { + previous := metrics.GlobalBlockReplay.AccountLoader + metrics.GlobalBlockReplay.AccountLoader = metrics.AccountLoader{} + return func() { + exec.accountLoader.Accumulate(metrics.GlobalBlockReplay.AccountLoader) + metrics.GlobalBlockReplay.AccountLoader = previous + } +} + +// loadTransactionAccounts resolves the group's address-table lookups and adds +// the pristine parent image of every account the group can touch to the +// parent snapshot, exactly as the whole-block loader does for a block, except +// that accounts already present keep their earlier image: the batch read at +// block.Slot through the same source returns parent state regardless of the +// overlay, so the first image is the right one and later groups must not +// replace it. +func (exec *blockExecution) loadTransactionAccounts(view *b.Block) error { + defer exec.captureAccountLoader()() + phaseStart := time.Now() + if err := resolveAddrTableLookups(exec.blockSrc, view); err != nil { + return fmt.Errorf("resolve address table lookups at slot %d: %w", view.Slot, err) + } + metrics.GlobalBlockReplay.AccountLoader.AddressTableLookups.AddTimingSince(phaseStart) + + phaseStart = time.Now() + dedupedAccts, _ := extractAndDedupeBlockAccts(view) + if exec.parentAccts.Map != nil { + filtered := dedupedAccts[:0] + for _, key := range dedupedAccts { + if _, loaded := exec.parentAccts.Map[key]; !loaded { + filtered = append(filtered, key) + } + } + dedupedAccts = filtered + } + metrics.GlobalBlockReplay.AccountLoader.DedupeBlockAccounts.AddTimingSince(phaseStart) + if len(dedupedAccts) == 0 { + return nil + } + + phaseStart = time.Now() + slotAccts, batchStats, err := getAccountsBatchSharedWithStats(context.Background(), exec.blockSrc, view.Slot, dedupedAccts) + metrics.GlobalBlockReplay.AccountLoader.SourceBatch.AddTimingSince(phaseStart) + recordAccountLoaderBatchStats(&metrics.GlobalBlockReplay.AccountLoader, batchStats) + if err != nil { + return fmt.Errorf("load transaction accounts at slot %d: %w", view.Slot, err) + } + if exec.parentAccts.Map == nil { + return fmt.Errorf("load transaction accounts at slot %d: parent snapshot is not a memory account set", view.Slot) + } + phaseStart = time.Now() + for _, acct := range slotAccts { + if acct == nil { + continue + } + if _, loaded := exec.parentAccts.Map[acct.Key]; loaded { + continue + } + key := [32]byte(acct.Key) + if err := exec.parentAccts.SetAccount(&key, acct); err != nil { + return err + } + } + metrics.GlobalBlockReplay.AccountLoader.ParentAccounts += uint64(len(slotAccts)) + metrics.GlobalBlockReplay.AccountLoader.ParentMapBuild.AddTimingSince(phaseStart) + return nil +} + +// runTransactionGroup is parallelTxLoop over a transaction slice with a plan +// built for the group alone (indices are group-local). Without the planner +// (txParallelism <= 1) it runs sequentially, which +// is always correct because groups are consumed in block order. +func (exec *blockExecution) runTransactionGroup(txs []*solana.Transaction, execute []bool, shouldVerifySignatures bool) ([]*fees.TxFeeInfo, []uint64, error) { + slotCtx := exec.slotCtx + feeInfos := make([]*fees.TxFeeInfo, len(txs)) + computeUnits := make([]uint64, len(txs)) + dbgOpts := exec.dbgOpts + + workers := exec.txParallelism + if workers > len(txs) { + workers = len(txs) + } + view := &b.Block{Transactions: txs} + if !canUseDependencyPlanner(view) { + return nil, nil, errors.New("streaming group has unresolved address tables") + } + var plan *dependencyPlan + if workers > 1 { + plannerBuildStart := time.Now() + plannerAccounts, _ := plannerAccountsForBlock(view) + plan = buildDependencyPlan(plannerAccounts) + metrics.GlobalBlockReplay.DependencyPlannerBuild.AddTimingSince(plannerBuildStart) + } + if plan == nil { + for idx, tx := range txs { + if !execute[idx] { + continue + } + var txErr error + feeInfos[idx], computeUnits[idx], txErr = ProcessTransaction(slotCtx, &exec.sigverifyWg, tx, nil, dbgOpts, nil, shouldVerifySignatures) + if feeInfos[idx] == nil { + return nil, nil, streamingTransactionError(idx, txErr) + } + } + return feeInfos, computeUnits, nil + } + + metrics.GlobalBlockReplay.DependencyPlannerPrepared = 1 + txErrors := make([]error, len(txs)) + do := make(chan int, len(txs)) + done := make(chan int, len(txs)) + plannerDone := make(chan struct{}) + go func() { + defer close(plannerDone) + plannerDispatchStart := time.Now() + dispatchDependencyPlan(plan, do, done) + metrics.GlobalBlockReplay.DependencyPlannerDispatch.AddTimingSince(plannerDispatchStart) + }() + + wg := &sync.WaitGroup{} + wg.Add(workers) + for i := 0; i < workers; i++ { + go func(workerIdx int) { + defer wg.Done() + var workerArena *arena.Arena[sealevel.BorrowedAccount] + if workerIdx < len(sealevel.BorrowedAccountArenas) { + workerArena = sealevel.BorrowedAccountArenas[workerIdx] + } + for idx := range do { + if !execute[idx] { + done <- idx + continue + } + feeInfos[idx], computeUnits[idx], txErrors[idx] = ProcessTransaction(slotCtx, &exec.sigverifyWg, txs[idx], nil, dbgOpts, workerArena, shouldVerifySignatures) + done <- idx + } + }(i) + } + wg.Wait() + close(done) + <-plannerDone + for idx := range txs { + if execute[idx] && feeInfos[idx] == nil { + return nil, nil, streamingTransactionError(idx, txErrors[idx]) + } + } + return feeInfos, computeUnits, nil +} + +// Instruction failures still carry charged fees and remain valid block entries. +// A missing fee result means transaction admission failed: discard the bank. +func streamingTransactionError(index int, err error) error { + if err == nil { + err = errors.New("missing fee result") + } + return fmt.Errorf("unprocessable streaming transaction %d: %w", index, err) +} + +// finalize runs the end-of-block phases over the open SlotCtx: fees to the +// leader, rent, incinerator, the Alpenglow footer clock and vote rewards, bank +// sysvar finalization, bank hash, footer verification, state publication and +// transaction status commit. It is the unchanged tail of ProcessBlock and is +// shared by streaming execution, which calls it once the complete block has +// been matched against the executed prefix. +func (exec *blockExecution) finalize() (*sealevel.SlotCtx, error) { + block := exec.block + slotCtx := exec.slotCtx + acctsDb := exec.acctsDb + tail := exec.tail + setReplayStage := exec.setReplayStage + alpenglowClock := exec.alpenglowClock + bankEpochSchedule := exec.bankEpochSchedule + txFeeAccumulator := exec.txFeeAccumulator + executionPlan := exec.executionPlan + transactionStatuses := exec.transactionStatuses + persistedHashes := exec.persistedHashes + var err error + + start := time.Now() + setReplayStage("distribute_fees") + + // distribute tx fees to the slot leader + // skip leader handling if there are zero transactions in this block + if !global.ManageLeaderSchedule() && block.BlockReward != nil && len(block.Transactions) > 0 { + slotCtx.LamportsBurnt = fees.DistributeTxFeesToSlotLeader(acctsDb, slotCtx, block.BlockReward.Leader, &txFeeAccumulator) + slotCtx.RecordModifiedAcct(block.BlockReward.Leader) + } else if global.ManageLeaderSchedule() && len(block.Transactions) > 0 { + slotCtx.LamportsBurnt = fees.DistributeTxFeesToSlotLeader(acctsDb, slotCtx, block.Leader, &txFeeAccumulator) + slotCtx.RecordModifiedAcct(block.Leader) + } + metrics.GlobalBlockReplay.Reward.AddTimingSince(start) + + start = time.Now() + setReplayStage("collect_rent") + bankRent, ok := slotCtx.BankSysvars().Rent() + if !ok { + return nil, fmt.Errorf("bank-local Rent sysvar unavailable at slot %d", block.Slot) + } + rentAccts := rent.CollectRentEagerly(slotCtx, &bankRent, bankEpochSchedule) + metrics.GlobalBlockReplay.Rent.AddTimingSince(start) + + start = time.Now() + setReplayStage("run_incinerator") + runIncinerator(slotCtx) + metrics.GlobalBlockReplay.RunIncinerator.AddTimingSince(start) + + // Alpenglow banks set the Clock timestamp from the block footer after execution. + if alpenglowClock { + footerClockStart := time.Now() + if err := applyAlpenglowFooterClock(slotCtx, block, bankEpochSchedule); err != nil { + metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) + return nil, fmt.Errorf("apply alpenglow footer clock at slot %d: %w", block.Slot, err) + } + if err := updateAlpenglowNanosecondClockAccount(slotCtx, block); err != nil { + metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) + return nil, err + } + metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) + voteRewardsStart := time.Now() + voteRewardsErr := ApplyAlpenglowVoteRewards(slotCtx, block, bankEpochSchedule, block.SkipRewardCert, block.NotarRewardCert, block.BlockFinalCert, block.AlpenglowShredVersion) + metrics.GlobalBlockReplay.AlpenglowVoteRewards.AddTimingSince(voteRewardsStart) + if voteRewardsErr != nil { + return nil, voteRewardsErr + } + } + if err := finalizeBankSysvars(slotCtx); err != nil { + return nil, fmt.Errorf("finalize bank sysvars at slot %d: %w", block.Slot, err) + } + + setReplayStage("compile_accounts") + start = time.Now() + writableAccts, modifiedAccts := compileWritableAndModifiedAccts(slotCtx, block, rentAccts) + metrics.GlobalBlockReplay.CompileWritableAndModifiedAccts.AddTimingSince(start) + start = time.Now() + ensureParentsErr := ensureParentAccountsForModified(slotCtx, modifiedAccts) + metrics.GlobalBlockReplay.EnsureParentAccountsForModified.AddTimingSince(start) + if ensureParentsErr != nil { + return nil, ensureParentsErr + } + + start = time.Now() + setReplayStage("bankhash") + slotCtx.FinalBankhash = bankhash.CalculateBankHash(slotCtx, writableAccts, modifiedAccts, block.ParentBankhash, slotCtx.NumSignatures, block.Blockhash) + metrics.GlobalBlockReplay.BankHash.AddTimingSince(start) + if alpenglowClock { + footerVerificationStart := time.Now() + footerVerificationErr := verifyAlpenglowBlockFooter(slotCtx, block, alpenglowClock) + metrics.GlobalBlockReplay.AlpenglowFooterVerification.AddTimingSince(footerVerificationStart) + if footerVerificationErr != nil { + writeFooterBankhashMismatchArtifact(footerVerificationErr, block, slotCtx, writableAccts, modifiedAccts) + return nil, footerVerificationErr + } + } + + // Bankhash consensus enforcement is handled in the replay loop (not here) + // because forkchoice is fed after ProcessBlock returns — checking here would + // never see votes from recently submitted blocks and could deadlock. + + // Enter critical commit window - panics here may leave AccountsDB inconsistent + commitSlot.Store(slotCtx.Slot) + commitInProgress.Store(true) + blockUpdateStart := time.Now() + setReplayStage("store_accounts") + persistedSlot := slotCtx.Slot + persistedBankhash := append([]byte(nil), slotCtx.FinalBankhash...) + persistedBlockSlot := block.Slot + stakeIndexDir := filepath.Join(acctsDb.AcctsDir, "..") + afterStoreAccounts := func() { + if tail != nil { + // Rooted-durable: accounts + bankhash are buffered in the overlay and + // become durable only on promotion; nothing written here (rooted-only). + } else { + if berr := acctsDb.StoreBankHashForSlot(persistedSlot, persistedBankhash); berr != nil { + mlog.Log.Infof("unable to store bankhash for slot %d", persistedSlot) + } + } + if tail == nil { + // Legacy/verify modes (no fork ambiguity): flush per block as before. + // Rooted-durable replay flushes at FOLD time instead — entries stay + // slot-scoped in RAM so a fork unwind can drop them, and scans merge + // the pending set (StreamStakeAccounts) for completeness meanwhile. + flushed, err := global.FlushPendingStakePubkeys(stakeIndexDir) + if err != nil { + mlog.Log.Errorf("failed to flush stake pubkey index: %v", err) + } else if flushed > 0 { + mlog.Log.Debugf("flushed %d new stake pubkeys to index", flushed) + } + } + + persistedHashes.Set(persistedBlockSlot, persistedBankhash) + + // Exit critical commit window - AccountsDB is now consistent + commitInProgress.Store(false) + commitSlot.Store(0) + } + + if tail != nil { + // Rooted-durable: buffer this slot's writes + bankhash in the RAM overlay + // (always, even when empty, so the bankhash is recorded); no durable write. + tail.Add(slotCtx.Slot, modifiedAccts, persistedBankhash) + afterStoreAccounts() + } else if len(modifiedAccts) > 0 { + err = acctsDb.StoreAccounts(modifiedAccts, slotCtx.Slot, afterStoreAccounts) + } + // In rooted-durable mode the callback above is synchronous, so this includes + // the complete critical-path overlay publication. Legacy StoreAccounts only + // enqueues here; its asynchronous disk work deliberately belongs to no slot's + // replay wall time and must never update a later slot's metrics record. + metrics.GlobalBlockReplay.BlockUpdateAccounts.AddTimingSince(blockUpdateStart) + if err != nil { + return slotCtx, err + } + statusCommitStart := time.Now() + statusWaitStart := time.Now() + preparedStatuses := exec.statusPreparation.wait() + metrics.GlobalBlockReplay.TransactionStatusPreparationWait.AddTimingSince(statusWaitStart) + statusErr := transactionStatuses.commitBlockWithValidation(block, executionPlan, preparedStatuses, exec.statusValidation) + metrics.GlobalBlockReplay.TransactionStatusCommit.AddTimingSince(statusCommitStart) + if statusErr != nil { + return nil, fmt.Errorf("commit transaction statuses for slot %d after bank state commit: %w", block.Slot, statusErr) + } + + global.IncrTransactionCount(executionPlan.processedTxCount) + setReplayStage("done") + return slotCtx, err +} diff --git a/pkg/replay/block_execution_test.go b/pkg/replay/block_execution_test.go new file mode 100644 index 000000000..31556fdf2 --- /dev/null +++ b/pkg/replay/block_execution_test.go @@ -0,0 +1,333 @@ +package replay + +import ( + "context" + "errors" + "math" + "math/rand" + "sort" + "sync" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/addresses" + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// memBlockSource serves a fixed parent state to the group loader the way the +// batch loader sees AccountsDB/the unrooted tail: a missing key yields a +// zero-lamport placeholder, never an error. +type memBlockSource struct { + mem accounts.MemAccounts +} + +func (s *memBlockSource) GetAccount(_ uint64, pubkey solana.PublicKey) (*accounts.Account, error) { + if acct, err := s.mem.GetAccountWithoutLock(pubkey); err == nil { + return acct.Clone(), nil + } + return &accounts.Account{Key: pubkey}, nil +} + +func (s *memBlockSource) GetAccountsBatch(_ context.Context, slot uint64, pks []solana.PublicKey) ([]*accounts.Account, error) { + out := make([]*accounts.Account, len(pks)) + for i, pk := range pks { + out[i], _ = s.GetAccount(slot, pk) + } + return out, nil +} + +// groupExecutionEnv is a block execution over a MemAccounts parent snapshot +// and overlay, mirroring what ProcessBlock builds through the account loader, +// with the process-wide sysvar cache set the way newCommitTestSlotCtx sets it. +type groupExecutionEnv struct { + exec *blockExecution + parent accounts.MemAccounts + source *memBlockSource + cleanup func() +} + +func newGroupExecutionEnv(t *testing.T, txParallelism int, payerLamports uint64) *groupExecutionEnv { + t.Helper() + feats := features.NewFeaturesDefault() + feats.EnableFeature(features.FormalizeLoadedTransactionDataSize, 0) + + durable := accounts.NewMemAccounts() + _ = durable.SetAccountWithoutLock(addresses.SystemProgramAddr, &accounts.Account{ + Key: addresses.SystemProgramAddr, Lamports: 1, Owner: addresses.NativeLoaderAddr, Executable: true, RentEpoch: math.MaxUint64, + }) + _ = durable.SetAccountWithoutLock(txfixture.PayerPubkey(), &accounts.Account{ + Key: txfixture.PayerPubkey(), Lamports: payerLamports, Owner: addresses.SystemProgramAddr, RentEpoch: math.MaxUint64, + }) + _ = durable.SetAccountWithoutLock(txfixture.DestPubkey(), &accounts.Account{ + Key: txfixture.DestPubkey(), Lamports: 10_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: math.MaxUint64, + }) + + prevRBH := sealevel.SysvarCache.RecentBlockHashes.Sysvar + rbh := sealevel.SysvarRecentBlockhashes{{Blockhash: txfixture.TestBlockhash(), FeeCalculator: sealevel.FeeCalculator{LamportsPerSignature: 5000}}} + sealevel.SysvarCache.RecentBlockHashes.Sysvar = &rbh + prevRent := sealevel.SysvarCache.Rent.Sysvar + rentSysvar := sealevel.NewDefaultRentSysvar() + sealevel.SysvarCache.Rent.Sysvar = &rentSysvar + + parent := accounts.NewMemAccounts() + overlay := accounts.NewOverlayAccounts(parent) + block := &b.Block{Slot: 42, Features: feats} + slotCtx := &sealevel.SlotCtx{ + Accounts: overlay, + ParentAccts: parent, + Slot: block.Slot, + Features: feats, + FeeRateGovernor: &sealevel.FeeRateGovernor{PrevLamportsPerSignature: 5000}, + LastBlockhash: txfixture.TestBlockhash(), + AcctMapsMu: &sync.Mutex{}, + ModifiedAccts: make(map[solana.PublicKey]bool), + WritableAccts: make(map[solana.PublicKey]bool), + VoteTimestampMu: &sync.Mutex{}, + VoteTimestamps: make(map[solana.PublicKey]sealevel.BlockTimestamp), + Replay: true, + } + source := &memBlockSource{mem: durable} + exec := &blockExecution{ + block: block, + txParallelism: txParallelism, + blockSrc: source, + slotCtx: slotCtx, + parentAccts: parent, + accts: overlay, + ctx: context.Background(), + setReplayStage: func(string) {}, + seenMessages: make(map[[32]byte]int), + } + return &groupExecutionEnv{ + exec: exec, + parent: parent, + source: source, + cleanup: func() { + sealevel.SysvarCache.RecentBlockHashes.Sysvar = prevRBH + sealevel.SysvarCache.Rent.Sysvar = prevRent + }, + } +} + +func (env *groupExecutionEnv) lamports(t *testing.T, key solana.PublicKey) uint64 { + t.Helper() + acct, err := env.exec.slotCtx.GetAccountShared(key) + require.NoError(t, err) + return acct.Lamports +} + +func (env *groupExecutionEnv) modifiedKeys() []string { + keys := make([]string, 0, len(env.exec.slotCtx.ModifiedAccts)) + for key := range env.exec.slotCtx.ModifiedAccts { + keys = append(keys, key.String()) + } + sort.Strings(keys) + return keys +} + +func transferTransactions(t *testing.T, n int, firstSeq uint64) []*solana.Transaction { + t.Helper() + txs := make([]*solana.Transaction, n) + for i := range txs { + tx, err := solana.TransactionFromBytes(txfixture.MustSignedTransferWire(firstSeq + uint64(i))) + require.NoError(t, err) + txs[i] = tx + } + return txs +} + +type groupExecutionOutcome struct { + payer, dest uint64 + fees uint64 + cu uint64 + processed, sigs uint64 + executeMask []bool + modified []string + parentPayerLamports uint64 +} + +func runGroups(t *testing.T, txParallelism int, payerLamports uint64, txs []*solana.Transaction, splits []int) groupExecutionOutcome { + t.Helper() + env := newGroupExecutionEnv(t, txParallelism, payerLamports) + defer env.cleanup() + start := 0 + for _, end := range append(append([]int(nil), splits...), len(txs)) { + if end < start { + end = start + } + require.NoError(t, env.exec.executeTransactionGroup(txs[start:end], nil, false)) + start = end + } + parentPayer, err := env.parent.GetAccountWithoutLock(txfixture.PayerPubkey()) + require.NoError(t, err) + return groupExecutionOutcome{ + payer: env.lamports(t, txfixture.PayerPubkey()), + dest: env.lamports(t, txfixture.DestPubkey()), + fees: env.exec.txFeeAccumulator.TotalFees, + cu: env.exec.totalCU, + processed: env.exec.processedTxCount, + sigs: env.exec.processedSignatures, + executeMask: append([]bool(nil), env.exec.execute...), + modified: env.modifiedKeys(), + parentPayerLamports: parentPayer.Lamports, + } +} + +// The reference is the pre-existing sequential path: ProcessTransaction over +// the block order on a slot context whose accounts were preloaded whole. +func runSequentialReference(t *testing.T, payerLamports uint64, txs []*solana.Transaction) groupExecutionOutcome { + t.Helper() + env := newGroupExecutionEnv(t, 0, payerLamports) + defer env.cleanup() + keys := []solana.PublicKey{addresses.SystemProgramAddr, txfixture.PayerPubkey(), txfixture.DestPubkey()} + for _, key := range keys { + acct, _ := env.source.GetAccount(42, key) + pk := [32]byte(key) + require.NoError(t, env.parent.SetAccount(&pk, acct)) + } + var sigverify sync.WaitGroup + var out groupExecutionOutcome + for i, tx := range txs { + feeInfo, cu, err := ProcessTransaction(env.exec.slotCtx, &sigverify, tx, nil, nil, nil, false) + // A failed transfer still returns its fee; a nil fee means the + // transaction was unprocessable (fee payer below rent exemption after + // the fee, bad blockhash...), which a valid block never contains. + require.NotNil(t, feeInfo, "transaction %d unprocessable: %v", i, err) + out.fees += feeInfo.TotalFee + out.cu += cu + out.processed++ + out.sigs += uint64(tx.Message.Header.NumRequiredSignatures) + out.executeMask = append(out.executeMask, true) + } + sigverify.Wait() + out.payer = env.lamports(t, txfixture.PayerPubkey()) + out.dest = env.lamports(t, txfixture.DestPubkey()) + out.modified = env.modifiedKeys() + out.parentPayerLamports = payerLamports + return out +} + +// TestExecuteTransactionGroupMatchesWholeBlock runs the same block-ordered +// transfers as one group, as random groups, and through the sequential +// reference. Every transfer shares the payer and destination, so every group +// boundary is a cross-group write dependency, and the payer balance is small +// enough that later transfers fail for insufficient funds, which exercises +// fee charging on failed transactions and makes outcomes order-dependent. +func TestExecuteTransactionGroupMatchesWholeBlock(t *testing.T) { + // Amounts are 999,001+ lamports each (seq%1e6+1) at a 5,000-lamport fee. + // With 3,200,000 lamports the first two transfers succeed (leaving + // 1,191,997) and the remaining 46 fail — the third would drop the payer + // below its 890,880-lamport rent-exempt minimum — while still paying + // their fee, ending at 961,997: every transaction stays processable (the + // fee payer never falls below rent exemption after the fee), which is + // what a valid block guarantees and what the loaders assert. + const payerLamports = 3_200_000 + txs := transferTransactions(t, 48, 999_000) + reference := runSequentialReference(t, payerLamports, txs) + require.Less(t, reference.payer, uint64(payerLamports)) + + for _, txParallelism := range []int{0, 1, 4} { + single := runGroups(t, txParallelism, payerLamports, txs, nil) + require.Equal(t, reference.payer, single.payer, "txpar %d single group payer", txParallelism) + require.Equal(t, reference.dest, single.dest, "txpar %d single group dest", txParallelism) + require.Equal(t, reference.fees, single.fees) + require.Equal(t, reference.cu, single.cu) + require.Equal(t, reference.processed, single.processed) + require.Equal(t, reference.sigs, single.sigs) + require.Equal(t, reference.executeMask, single.executeMask) + require.Equal(t, reference.modified, single.modified) + require.Equal(t, uint64(payerLamports), single.parentPayerLamports, "parent image must stay pristine") + + rng := rand.New(rand.NewSource(int64(7 + txParallelism))) + for iter := 0; iter < 8; iter++ { + splitCount := 1 + rng.Intn(6) + splits := make([]int, splitCount) + for i := range splits { + splits[i] = rng.Intn(len(txs) + 1) + } + sort.Ints(splits) + grouped := runGroups(t, txParallelism, payerLamports, txs, splits) + require.Equal(t, single, grouped, "txpar %d splits %v", txParallelism, splits) + } + } +} + +func TestExecuteTransactionGroupRejectsDuplicatesAcrossGroups(t *testing.T) { + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + defer env.cleanup() + txs := transferTransactions(t, 3, 100) + require.NoError(t, env.exec.executeTransactionGroup(txs[:2], nil, false)) + err := env.exec.executeTransactionGroup([]*solana.Transaction{txs[2], txs[0]}, nil, false) + var duplicateErr *DuplicateTransactionMessagesError + require.Error(t, err) + require.True(t, errors.As(err, &duplicateErr)) + require.Equal(t, uint64(42), duplicateErr.Slot) + require.Equal(t, uint64(1), duplicateErr.DuplicateCount) + require.Equal(t, []DuplicateTransactionOccurrence{{Index: 3, FirstIndex: 0}}, duplicateErr.Occurrences) + // The rejected group must not have been recorded or executed. + require.Len(t, env.exec.transactions, 2) + require.Equal(t, uint64(2), env.exec.processedTxCount) +} + +func TestExecuteTransactionGroupRejectsV1BeforeActivation(t *testing.T) { + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + defer env.cleanup() + tx, err := solana.TransactionFromBytes(txfixture.MustSignedV1Wire(9, 8)) + require.NoError(t, err) + err = env.exec.executeTransactionGroup([]*solana.Transaction{tx}, nil, false) + require.ErrorIs(t, err, TxErrUnsupportedVersion) + require.Empty(t, env.exec.transactions) +} + +func TestExecuteTransactionGroupKeepsFirstParentImage(t *testing.T) { + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + defer env.cleanup() + txs := transferTransactions(t, 4, 200) + require.NoError(t, env.exec.executeTransactionGroup(txs[:2], nil, false)) + afterFirst := env.lamports(t, txfixture.PayerPubkey()) + require.Less(t, afterFirst, uint64(10_000_000_000)) + // A later group touching the same accounts must not reload the payer's + // parent image over the pristine one, nor see stale overlay state. + require.NoError(t, env.exec.executeTransactionGroup(txs[2:], nil, false)) + parentPayer, err := env.parent.GetAccountWithoutLock(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Equal(t, uint64(10_000_000_000), parentPayer.Lamports) + require.Less(t, env.lamports(t, txfixture.PayerPubkey()), afterFirst) + require.Equal(t, 2, env.exec.groups) + require.Len(t, env.exec.transactions, 4) +} + +func TestExecuteTransactionGroupRefusesClosedExecution(t *testing.T) { + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + defer env.cleanup() + env.exec.closed = true + err := env.exec.executeTransactionGroup(transferTransactions(t, 1, 300), nil, false) + require.ErrorIs(t, err, errBlockExecutionClosed) +} + +func TestStreamingGroupRejectsUnprocessableTransactions(t *testing.T) { + for _, workers := range []int{0, 4} { + env := newGroupExecutionEnv(t, workers, 10_000_000) + defer env.cleanup() + txs := transferTransactions(t, 2, 1) + txs[0].Message.RecentBlockhash = solana.Hash{0xFA} + err := env.exec.executeTransactionGroup(txs, nil, false) + require.ErrorIs(t, err, TxErrInvalidBlockhash) + } +} + +func TestStreamingGroupRejectsUnresolvedLookupsBeforeExecution(t *testing.T) { + for _, workers := range []int{0, 4} { + env := newGroupExecutionEnv(t, workers, 10_000_000) + defer env.cleanup() + txs := decodeTransactions(t, [][]byte{signedV0TransferViaTableWire(t, 1)}) + err := env.exec.executeTransactionGroup(txs, nil, false) + require.ErrorContains(t, err, "unresolved address tables") + require.Zero(t, env.exec.slotCtx.TotalComputeUnitsConsumed) + } +} diff --git a/pkg/replay/chain_state.go b/pkg/replay/chain_state.go index 649146c93..70c404b22 100644 --- a/pkg/replay/chain_state.go +++ b/pkg/replay/chain_state.go @@ -266,3 +266,14 @@ func ChainTipFeatureActive(gate features.FeatureGate) bool { defer chainTipMu.RUnlock() return chainTipFeatures != nil && chainTipFeatures.IsActive(gate) } + +// ChainTipFeatures returns an independent feature snapshot for queued transaction +// preparation. Bank admission checks compatibility again before reuse. +func ChainTipFeatures() *features.Features { + chainTipMu.RLock() + defer chainTipMu.RUnlock() + if chainTipFeatures == nil { + return nil + } + return chainTipFeatures.Clone() +} diff --git a/pkg/replay/promotion.go b/pkg/replay/promotion.go index c750009ea..5debf2ffa 100644 --- a/pkg/replay/promotion.go +++ b/pkg/replay/promotion.go @@ -27,14 +27,13 @@ type batchCommitter interface { CommitBatch(deltas []accounts.SlotDelta, throughSlot uint64, bankhashes map[uint64][32]byte, resumeCtx []byte) (accountsdb.BatchCommitResult, error) } -// TransactionStatusCheckpointHooks deliberately split status-cache capture -// from sidecar I/O. Snapshot runs on the replay loop while its mutable cache is -// coherent; Install runs on the fold worker using only those immutable bytes. -// This makes it impossible for the async worker to traverse concurrently -// changing replay lineage. The later AccountsDB manifest remains the selector. +// TransactionStatusCheckpointHooks split immutable status capture from encoding +// and sidecar I/O. Capture runs on replay; the fold worker serializes the captured +// view and then calls Install. Neither worker operation revisits live lineage. +// The later AccountsDB manifest remains the durable checkpoint selector. type TransactionStatusCheckpointHooks struct { - Snapshot func(through uint64) ([]byte, error) - Install func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) + Capture func(through uint64) (TransactionStatusSnapshot, error) + Install func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) // AfterCommit is an advisory retention hook. It runs only after CommitBatch // has durably selected the manifest carrying selected. Its error is logged // and ignored: once CommitBatch succeeds, the fold must remain successful. @@ -378,7 +377,10 @@ type foldJob struct { ctx *state.ResumeContext ctxJSON []byte stakeIdxDir string - transactionStatusCheckpointPayload []byte + transactionStatusSnapshot TransactionStatusSnapshot + checkpointCaptureTime time.Duration + checkpointEncodeTime time.Duration + checkpointBytes int installTransactionStatusCheckpoint func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) afterTransactionStatusCheckpointCommit func(selected *state.TransactionStatusCheckpointRef) error } @@ -388,6 +390,23 @@ type foldResult struct { err error } +// buildRewardsCompletionFoldJob puts the entire rewards window in one commit. +// A normal batch cutoff inside that window would leave an active EpochRewards +// checkpoint without the RAM-only spool bookkeeping needed to resume it. The +// retained tail already bounds the size of this once-per-epoch fold. +func (t *unrootedTail) buildRewardsCompletionFoldJob(through uint64) (*foldJob, error) { + whole := *t + whole.batchSlots = t.overlay.HeldSlots() + job, err := whole.buildFoldJob(through, true) + if err != nil { + return nil, err + } + if job == nil || job.through != through { + return nil, fmt.Errorf("rewards completion bank %d is absent from retained fold prefix", through) + } + return job, nil +} + // buildFoldJob snapshots the FIRST fold chunk of the rooted prefix <= through // (loop thread). force also takes a trailing partial chunk. Returns nil when // no chunk is ready. A missing chunk-top context is an error — a context-less @@ -398,16 +417,13 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. if err != nil { return nil, err } - prefix := t.overlay.PromotionPrefix(through) - if len(prefix) == 0 { + // Check chunk eligibility before materializing account-write lists. Replay + // calls this on every iteration, including skipped slots; a partial batch + // remains in RAM without rescanning all of its accounts each time. + chunk := t.overlay.PromotionChunk(through, t.batchSlots, force) + if len(chunk) == 0 { return nil, nil } - chunk := prefix - if len(chunk) > t.batchSlots { - chunk = chunk[:t.batchSlots] - } else if len(chunk) < t.batchSlots && !force { - return nil, nil // trailing partial chunk stays in RAM - } through = chunk[len(chunk)-1].Slot ctx := t.contexts[through] @@ -415,18 +431,18 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. return nil, fmt.Errorf("fold chunk through slot %d: no resume context recorded for chunk-top slot", through) } ctx = cloneResumeContextForFold(ctx) - var checkpointPayload []byte - if hooks.Snapshot != nil { - checkpointPayload, err = hooks.Snapshot(through) + var snapshot TransactionStatusSnapshot + var captureTime time.Duration + if hooks.Capture != nil { + start := time.Now() + snapshot, err = hooks.Capture(through) + captureTime = time.Since(start) if err != nil { - return nil, fmt.Errorf("fold chunk through slot %d: snapshot transaction status checkpoint: %w", through, err) + return nil, fmt.Errorf("fold chunk through slot %d: capture transaction status checkpoint: %w", through, err) } - if len(checkpointPayload) == 0 { - return nil, fmt.Errorf("fold chunk through slot %d: transaction status checkpoint snapshot is empty", through) + if snapshot == nil { + return nil, fmt.Errorf("fold chunk through slot %d: transaction status checkpoint capture is nil", through) } - // The worker owns this immutable copy. Even a future Snapshot - // implementation that reuses a scratch buffer cannot race it. - checkpointPayload = append([]byte(nil), checkpointPayload...) } bankhashes := make(map[uint64][32]byte, len(chunk)) for _, sd := range chunk { @@ -440,7 +456,8 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. bankhashes: bankhashes, ctx: ctx, stakeIdxDir: t.stakeIdxDir, - transactionStatusCheckpointPayload: checkpointPayload, + transactionStatusSnapshot: snapshot, + checkpointCaptureTime: captureTime, installTransactionStatusCheckpoint: hooks.Install, afterTransactionStatusCheckpointCommit: hooks.AfterCommit, }, nil @@ -450,12 +467,26 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. // state). Stake-index entries flush (fsync'd) BEFORE the batch commit — see // promoteRootedBatched for why that order is a correctness requirement. func runFoldJob(committer batchCommitter, job *foldJob) error { - if job == nil || job.ctx == nil { + if job == nil { + return errors.New("fold job has no resume context") + } + // Failed folds are rebuilt from the retained tail. Neither a failed result + // nor a completed-but-unapplied job should keep checkpoint deltas alive. + defer func() { job.transactionStatusSnapshot = nil }() + if job.ctx == nil { return errors.New("fold job has no resume context") } var selectedCheckpoint *state.TransactionStatusCheckpointRef if job.installTransactionStatusCheckpoint != nil { - ref, err := job.installTransactionStatusCheckpoint(job.through, job.transactionStatusCheckpointPayload) + start := time.Now() + payload, err := encodeTransactionStatusCheckpoint(job.transactionStatusSnapshot) + job.checkpointEncodeTime = time.Since(start) + job.transactionStatusSnapshot = nil + if err != nil { + return fmt.Errorf("fold chunk through slot %d: encode transaction status checkpoint: %w", job.through, err) + } + job.checkpointBytes = len(payload) + ref, err := job.installTransactionStatusCheckpoint(job.through, payload) if err != nil { return fmt.Errorf("fold chunk through slot %d: prepare transaction status checkpoint: %w", job.through, err) } @@ -539,7 +570,9 @@ func (p *asyncPromoter) run() { start := time.Now() err := runFoldJob(p.committer, job) if err == nil { - mlog.Log.FileOnlyf("async fold: committed %d slots through %d in %s", len(job.chunk), job.through, time.Since(start).Round(time.Millisecond)) + mlog.Log.FileOnlyf("async fold: committed %d slots through %d in %s checkpoint_capture=%s checkpoint_encode=%s checkpoint_bytes=%d", + len(job.chunk), job.through, time.Since(start).Round(time.Millisecond), + job.checkpointCaptureTime, job.checkpointEncodeTime, job.checkpointBytes) } p.results <- foldResult{job: job, err: err} } @@ -594,13 +627,11 @@ func (p *asyncPromoter) stop() { // the caller validates the pair and falls back to rooted-checkpoint re-replay. func (t *unrootedTail) unwind(fromSlot uint64) (*state.ResumeContext, *sealevel.BankSysvars) { t.overlay.EvictFrom(fromSlot) - // Branch-scoped side effect: stake pubkeys enqueued by the evicted slots - // must never reach the durable index — drop them with the state. - if dropped := global.DropPendingStakePubkeysFrom(fromSlot); dropped > 0 { - mlog.Log.Infof("fork unwind: dropped %d pending stake-index entries from slots >= %d", dropped, fromSlot) - } + // Only replay-owned held slots are unwound. A future local leader bank + // is not part of this tail and must retain its pending stake entries. for s := range t.bankhashes { if s >= fromSlot { + global.DropPendingStakePubkeys(s) delete(t.bankhashes, s) } } @@ -713,14 +744,15 @@ func promoteRootedBatched( } ctx = cloneResumeContextForFold(ctx) var selectedCheckpoint *state.TransactionStatusCheckpointRef - if hooks.Snapshot != nil { - payload, serr := hooks.Snapshot(chunkThrough) + if hooks.Capture != nil { + snapshot, serr := hooks.Capture(chunkThrough) if serr != nil { - err = fmt.Errorf("promote chunk through slot %d: snapshot transaction status checkpoint: %w", chunkThrough, serr) + err = fmt.Errorf("promote chunk through slot %d: capture transaction status checkpoint: %w", chunkThrough, serr) break } - if len(payload) == 0 { - err = fmt.Errorf("promote chunk through slot %d: transaction status checkpoint snapshot is empty", chunkThrough) + payload, serr := encodeTransactionStatusCheckpoint(snapshot) + if serr != nil { + err = fmt.Errorf("promote chunk through slot %d: encode transaction status checkpoint: %w", chunkThrough, serr) break } ref, perr := hooks.Install(chunkThrough, payload) @@ -792,15 +824,29 @@ func resolveTransactionStatusCheckpointHooks(configured TransactionStatusCheckpo } func validateTransactionStatusCheckpointHooks(hooks TransactionStatusCheckpointHooks) error { - if (hooks.Snapshot == nil) != (hooks.Install == nil) { - return errors.New("transaction status checkpoint Snapshot and Install hooks must either both be set or both be nil") + if (hooks.Capture == nil) != (hooks.Install == nil) { + return errors.New("transaction status checkpoint Capture and Install hooks must either both be set or both be nil") } if hooks.AfterCommit != nil && hooks.Install == nil { - return errors.New("transaction status checkpoint AfterCommit hook requires Snapshot and Install hooks") + return errors.New("transaction status checkpoint AfterCommit hook requires Capture and Install hooks") } return nil } +func encodeTransactionStatusCheckpoint(snapshot TransactionStatusSnapshot) ([]byte, error) { + if snapshot == nil { + return nil, errors.New("transaction status checkpoint capture is nil") + } + payload, err := snapshot.MarshalBinary() + if err != nil { + return nil, err + } + if len(payload) == 0 { + return nil, errors.New("transaction status checkpoint snapshot is empty") + } + return payload, nil +} + func cloneResumeContextForFold(ctx *state.ResumeContext) *state.ResumeContext { if ctx == nil { return nil diff --git a/pkg/replay/remaining_compute_units_test.go b/pkg/replay/remaining_compute_units_test.go index ddec0e686..2cf0ace1c 100644 --- a/pkg/replay/remaining_compute_units_test.go +++ b/pkg/replay/remaining_compute_units_test.go @@ -77,7 +77,9 @@ func TestRemainingComputeUnitsPreservesSuccessfulNonceAdvance(t *testing.T) { } program := &sbpf.Program{Text: text, TextBytes: textBytes, TextVA: sbpf.VaddrProgram} require.NoError(t, program.Verify()) - slotCtx.AccountsDb.AddProgramToCache(programKey, &accountsdb.ProgramCacheEntry{Program: program}) + entry := &accountsdb.ProgramCacheEntry{Program: program} + entry.BindSource(nil, slotCtx.Features) + slotCtx.AccountsDb.AddProgramToCache(programKey, entry) tx, err := solana.NewTransaction([]solana.Instruction{ system.NewAdvanceNonceAccountInstruction(nonceKey, solana.SysVarRecentBlockHashesPubkey, payer).Build(), diff --git a/pkg/replay/rewards_retirement.go b/pkg/replay/rewards_retirement.go new file mode 100644 index 000000000..18783e382 --- /dev/null +++ b/pkg/replay/rewards_retirement.go @@ -0,0 +1,62 @@ +package replay + +import ( + "github.com/Overclock-Validator/mithril/pkg/rewards" + "github.com/Overclock-Validator/mithril/pkg/sealevel" +) + +// partitionedRewardsCompletion is replay-thread-owned, process-local evidence +// that a successfully executed bank contains all effects of this distribution. +// It is not a checkpoint or signing authority. Until that bank is durable, +// tryInLoopUnwind must still reject even a zero-remaining distribution: its +// spool has been consumed and cannot be rolled back with the account overlay. +type partitionedRewardsCompletion struct { + info *rewards.PartitionedRewardDistributionInfo + slot uint64 +} + +// limitPromotion keeps the boundary replayable until a successfully verified +// completion bank is eligible for promotion. Consuming the last spool changes +// the RAM counter before footer verification and is not completion evidence. +func (c *partitionedRewardsCompletion) limitPromotion(info *rewards.PartitionedRewardDistributionInfo, boundary, through uint64) uint64 { + if boundary == 0 || info == nil { + return through + } + if info.NumRewardPartitionsRemaining != 0 || c.info != info || c.slot == 0 || through < c.slot { + return min(through, boundary-1) + } + return through +} + +// observeBank must run only after successful block execution/publication, using +// that bank's immutable sysvars (never the speculative global sysvar cache). +// If the first completed bank lacks evidence, recording a later descendant is +// conservative: retirement then waits for that later bank to become durable. +func (c *partitionedRewardsCompletion) observeBank(info *rewards.PartitionedRewardDistributionInfo, bank *sealevel.BankSysvars) { + if c.info != info { + *c = partitionedRewardsCompletion{info: info} + } + if info == nil || c.slot != 0 || info.NumRewardPartitionsRemaining != 0 || bank == nil || bank.Slot() == 0 { + return + } + epochRewards, ok := bank.EpochRewards() + if ok && !epochRewards.Active { + c.slot = bank.Slot() + } +} + +// retire is called only when replay applies a successfully committed fold and +// advances LastRootedSlot. Finality, an enqueued/in-flight fold, and a failed +// commit do not acknowledge durability. At this boundary every rewards effect +// is in AccountsDB; in-memory switches above it cannot undo distribution. +// Switches at/below it still take durable recovery, whose persisted +// EpochRewards validation remains unchanged. Restart loses this optional +// evidence and reconstructs state through the existing recovery path. +func (c *partitionedRewardsCompletion) retire(info **rewards.PartitionedRewardDistributionInfo, durableSlot uint64) bool { + if *info == nil || *info != c.info || c.slot == 0 || durableSlot < c.slot || (*info).NumRewardPartitionsRemaining != 0 { + return false + } + *info = nil + *c = partitionedRewardsCompletion{} + return true +} diff --git a/pkg/replay/rewards_retirement_test.go b/pkg/replay/rewards_retirement_test.go new file mode 100644 index 000000000..aaa169ae1 --- /dev/null +++ b/pkg/replay/rewards_retirement_test.go @@ -0,0 +1,182 @@ +package replay + +import ( + "bytes" + "encoding/base64" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/rewards" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/state" + bin "github.com/gagliardetto/binary" + "github.com/mr-tron/base58" + "github.com/stretchr/testify/require" +) + +func TestRewardsRetirementRequiresCompletedBankAndDurability(t *testing.T) { + active := &sealevel.SysvarEpochRewards{Active: true} + var raw bytes.Buffer + require.NoError(t, active.MarshalWithEncoder(bin.NewBinEncoder(&raw))) + activeBank, err := sealevel.NewBankSysvars(5, &accounts.Account{Key: sealevel.SysvarEpochRewardsAddr, Data: raw.Bytes()}) + require.NoError(t, err) + missingBank, err := sealevel.NewBankSysvars(5) + require.NoError(t, err) + for _, tc := range []struct { + name string + remaining uint64 + bank *sealevel.BankSysvars + }{ + {"active distribution", 1, testUnwindBankSysvars(t, 5, 50)}, + {"active bank", 0, activeBank}, + {"missing bank", 0, nil}, + {"missing rewards", 0, missingBank}, + {"unknown slot", 0, testUnwindBankSysvars(t, 0, 50)}, + } { + t.Run(tc.name, func(t *testing.T) { + info := &rewards.PartitionedRewardDistributionInfo{NumRewardPartitionsRemaining: tc.remaining} + var completed partitionedRewardsCompletion + completed.observeBank(info, tc.bank) + require.False(t, completed.retire(&info, 100)) + require.NotNil(t, info) + }) + } + + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + require.False(t, completed.retire(&info, 100), "zero remaining without observed completion is insufficient") + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + completed.observeBank(info, testUnwindBankSysvars(t, 7, 50)) + require.False(t, completed.retire(&info, 4), "uncommitted completion must retain the guard") + require.True(t, completed.retire(&info, 5), "later observations must not postpone recorded completion") + require.Nil(t, info) + require.False(t, completed.retire(&info, 100), "retirement is one-shot") +} + +func TestRewardsPromotionRetainsBoundaryAfterFailedDistribution(t *testing.T) { + const boundary = uint64(6264000) + info := &rewards.PartitionedRewardDistributionInfo{NumRewardPartitionsRemaining: 1} + var completed partitionedRewardsCompletion + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary+1)) + + // Distribution consumes the final spool before ProcessBlock checks the + // footer. The epoch-116 failure took this path; no successful bank was + // observed, so forced shutdown must not persist the boundary bank. + info.NumRewardPartitionsRemaining = 0 + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary+1)) + completed.observeBank(info, nil) + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary+1)) + + completed.observeBank(info, testUnwindBankSysvars(t, boundary+1, 50)) + require.Equal(t, boundary-2, completed.limitPromotion(info, boundary, boundary-2)) + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary), "finality stopped inside rewards window") + require.Equal(t, boundary+1, completed.limitPromotion(info, boundary, boundary+1)) + + next := &rewards.PartitionedRewardDistributionInfo{} + require.Equal(t, boundary-1, completed.limitPromotion(next, boundary, boundary+2), "completion belongs to another distribution") + require.Equal(t, boundary+2, completed.limitPromotion(nil, boundary, boundary+2)) + require.Equal(t, boundary+2, completed.limitPromotion(info, 0, boundary+2)) +} + +func TestRewardsCompletionFoldCannotCheckpointInsideWindow(t *testing.T) { + for _, batchSize := range []int{1, 2, 128} { + t.Run(fmt.Sprintf("batch_%d", batchSize), func(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts(), failOn: 7} + tail := asyncTestTail(fc, 5, 6, 7, 8) + tail.batchSlots = batchSize + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 7, 50)) + through := completed.limitPromotion(info, 5, 8) + require.Equal(t, uint64(8), through) + + job, err := tail.buildRewardsCompletionFoldJob(completed.slot) + require.NoError(t, err) + require.NotNil(t, job) + require.Equal(t, uint64(7), job.through) + require.Len(t, job.chunk, 3, "the entire distribution must share one commit") + require.Equal(t, batchSize, tail.batchSlots, "normal batching is unchanged") + require.Error(t, runFoldJob(fc, job)) + require.Empty(t, fc.throughs) + require.False(t, completed.retire(&info, 4)) + + fc.failOn = 0 + job, err = tail.buildRewardsCompletionFoldJob(completed.slot) + require.NoError(t, err) + require.NoError(t, runFoldJob(fc, job)) + require.Equal(t, []uint64{7}, fc.throughs) + tail.applyFoldJob(job) + require.True(t, completed.retire(&info, 7)) + require.Equal(t, 1, tail.overlay.HeldSlots(), "later banks remain buffered") + }) + } + tail := asyncTestTail(&fakeCommitter{durable: accounts.NewMemAccounts()}, 5, 6) + _, err := tail.buildRewardsCompletionFoldJob(7) + require.ErrorContains(t, err, "absent from retained fold prefix") +} + +func TestRewardsRetirementDoesNotCrossGenerations(t *testing.T) { + old := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 1} + next := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 10} + var completed partitionedRewardsCompletion + completed.observeBank(old, testUnwindBankSysvars(t, 5, 50)) + require.False(t, completed.retire(&next, 100), "old completion cannot retire new bookkeeping") + completed.observeBank(next, testUnwindBankSysvars(t, 11, 60)) + require.False(t, completed.retire(&next, 10)) + require.True(t, completed.retire(&next, 11)) +} + +func TestRewardsRetirementWaitsForSuccessfulFold(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts(), failOn: 5} + tail := asyncTestTail(fc, 5, 6) + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + job, err := tail.buildFoldJob(6, true) + require.NoError(t, err) + require.NotNil(t, job) + root := uint64(4) + require.False(t, completed.retire(&info, root), "capturing a job does not make its bank durable") + require.Error(t, runFoldJob(fc, job)) + require.False(t, completed.retire(&info, root), "a failed fold leaves the old durable root") + fc.failOn = 0 + require.NoError(t, runFoldJob(fc, job)) + ctx := tail.applyFoldJob(job) + require.NotNil(t, ctx) + root = job.through + require.True(t, completed.retire(&info, root)) +} + +func TestRewardsRetirementAllowsExactParentUnwind(t *testing.T) { + resetVoteStakeDirty() + t.Cleanup(resetVoteStakeDirty) + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + tail := newUnrootedTail(&fakeDurable{}, &fakeCommitter{durable: accounts.NewMemAccounts()}, 512, 1, "") + parent := &state.ResumeContext{Slot: 7, Bankhash: base58.Encode(make([]byte, 32)), AcctsLtHash: base64.StdEncoding.EncodeToString(make([]byte, 2048)), Capitalization: 700} + bank := testUnwindBankSysvars(t, 7, 50) + tail.Add(7, []*accounts.Account{testAccount(1, 71)}, testHashBytes(7)) + tail.SetContext(7, parent, bank) + tail.Add(8, []*accounts.Account{testAccount(1, 81)}, testHashBytes(8)) + tail.SetContext(8, &state.ResumeContext{Slot: 8}, testUnwindBankSysvars(t, 8, 999)) + sw := &CertifiedSwitch{Slot: 8} + ms := &state.MithrilState{LastRootedSlot: 4} + sched := &sealevel.SysvarEpochSchedule{SlotsPerEpoch: 432000} + rs, _, reason := tryInLoopUnwind(sw, tail, ms, sched, 0, info) + require.Nil(t, rs) + require.Equal(t, unwindFallbackRewardsWindow, reason) + ms.LastRootedSlot = 5 + markVoteStakeDirty(5) // completed reward writes are also below the durable root + require.True(t, completed.retire(&info, ms.LastRootedSlot)) + rs, restored, reason := tryInLoopUnwind(sw, tail, ms, sched, 0, info) + require.Empty(t, reason) + require.Same(t, bank, restored, "use the surviving bank, never abandoned reward sysvars") + want, err := ResumeStateFromRootedContext(parent, nil) + require.NoError(t, err) + require.Equal(t, want, rs, "resume state must match rebuilding the exact retained parent") + acct, err := tail.GetAccount(8, testAccount(1, 0).Key) + require.NoError(t, err) + require.Equal(t, uint64(71), acct.Lamports, "abandoned account writes must be removed") +} diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go new file mode 100644 index 000000000..2a02820bb --- /dev/null +++ b/pkg/replay/streaming.go @@ -0,0 +1,1230 @@ +package replay + +import ( + "context" + "errors" + "fmt" + "maps" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/blockstream" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// Streaming execution executes a turbine block's entry batches while the rest +// of its shreds are still arriving, so that only the last batch and the +// end-of-block tail remain after the final shred. The complete block from the +// ordinary emission path stays the authority: the executor only pre-computes +// the bank overlay for a prefix of it, proves at finalize that the prefix is +// the block (pointer identity of every executed transaction, same parent slot +// and ID, same generation, same features), and otherwise throws the prefix +// away and lets the whole-block path execute the block from scratch. +// +// Nothing a stream does reaches process-global state until finalize: the +// overlay lives in the SlotCtx, vote-cache publication is deferred on the +// SlotCtx, program-cache insertions are recorded for undo, pending stake index +// entries are slot-keyed and dropped, the parent's VoteTimestamps map is +// cloned at open, the process-wide current slot is not published for the +// shell, and the legacy sysvar cache written by bank open is snapshotted and +// restored. Discard therefore restores the world to the state before the +// stream opened. + +// StreamingExecutionConfig is set from the node flags before replay starts. +type StreamingExecutionConfig struct { + // Enabled turns streaming execution on for turbine-sourced blocks. + Enabled bool + // Workers bounds the executor goroutines per group (0 = min(txpar, 4)). + Workers int + // MinGroupBatches delays a group until this many contiguous batches are + // ready, unless the slot is already complete (0 or 1 = execute as soon as + // one batch is ready). + MinGroupBatches int + // MaxOpenAge discards a stream that has been open this long without its + // block completing (0 = 2 s). + MaxOpenAge time.Duration +} + +// StreamingExecutionCfg is the process-wide streaming configuration. +var StreamingExecutionCfg StreamingExecutionConfig + +const ( + defaultStreamingWorkers = 4 + defaultStreamingMaxAge = 2 * time.Second + streamingPollInterval = 5 * time.Millisecond + // Bound the entire group's verification join, not each batch separately. + // This is a speculative-work budget, not a signature validity deadline. + streamingVerificationWait = 100 * time.Millisecond + // Bound speculation across missing leaders; this never advances replay + // or establishes that the intervening slots are actually skipped. + streamingMaxSlotDistance = uint64(32) + // streamingHardOpenAgeFactor bounds a completed-but-not-yet-emitted stream + // to this multiple of MaxOpenAge. + streamingHardOpenAgeFactor = 10 +) + +func (cfg StreamingExecutionConfig) workers(txParallelism int) int { + workers := cfg.Workers + if workers <= 0 { + workers = defaultStreamingWorkers + } + if txParallelism > 0 && workers > txParallelism { + workers = txParallelism + } + return workers +} + +func (cfg StreamingExecutionConfig) maxOpenAge() time.Duration { + if cfg.MaxOpenAge <= 0 { + return defaultStreamingMaxAge + } + return cfg.MaxOpenAge +} + +// streamingFeed is the block source's view of the turbine feed +// (*blockstream.BlockSource implements it; tests substitute a fake). +type streamingFeed interface { + StreamEvents() <-chan turbine.StreamEvent + StreamStatusOf(turbine.StreamGeneration) turbine.StreamStatus + PendingStreamBatches(turbine.StreamGeneration, uint32) []*turbine.StreamBatch + PrioritizeStreamRepair(turbine.StreamGeneration) +} + +var _ streamingFeed = (*blockstream.BlockSource)(nil) + +// streamingDeps is what the executor needs from the replay loop. The closures +// read loop-local state (last slot context, frontier, features, switch +// status) at call time so the executor never caches a stale view. +type streamingDeps struct { + acctsDb *accountsdb.AccountsDb + feed streamingFeed + epochSchedule *sealevel.SysvarEpochSchedule + txParallelism int + dbgOpts *DebugOptions + persistedHashes *persistedTracker + tail unrootedState + transactionStatuses *TransactionStatusCache + alpenglowClock bool + + lastSlotCtx func() *sealevel.SlotCtx + frontier func() uint64 + // frontierMark reports the last executed block's replay and full instants + // (timeline only; nil when the loop does not track it). + frontierMark func() streamingFrontierMark + currentFeatures func() *features.Features + currentEpoch func() uint64 + rewardsInFlight func() bool + switchPending func() bool + executedBlockID func(slot uint64) (solana.Hash, bool) + alpenglowMode bool + unrootedTailUsed bool +} + +type streamingGroup struct { + // readyAt is when the group was formed from contiguous decoded batches; + // joinedAt when the consumer finished joining verification and copying + // batch slices (not the verifier completion instant); startedAt and + // finishedAt bound the execution itself. + readyAt, joinedAt, startedAt, finishedAt time.Time + batches, transactions int + suffix bool // the finalize suffix, run on the complete block +} + +// streamingFrontierMark is the replay loop's record of the last executed +// block: the slot, the instant its replay result reached consensus, and its +// last-shred instant (0 for a block that did not arrive as shreds). Skips do +// not update it. It only feeds the timeline; the executor never decides +// anything on it. +type streamingFrontierMark struct { + slot uint64 + fullNanos int64 + // admittedAt is when the source handed the block to replay (after its + // ancestors were replayed and the emitter released it); replayedAt when + // its replay result reached consensus. + admittedAt time.Time + replayedAt time.Time + // waitEnteredAt is the loop's first entry into the replay wait after + // replayedAt; what lies between is the executed block's post-replay tail. + waitEnteredAt time.Time +} + +// streamingObservation is what the executor knows about a slot's header. It +// outlives the header itself (pruned when the frontier passes the slot) so +// that a block executed whole can report why no stream opened for it, and a +// stream can report how long its header waited and on what. +type streamingObservation struct { + verificationWait metrics.Timing + generation turbine.StreamGeneration + parentSlot uint64 + readyAt time.Time // header batch decoded (its wake-up's ReadyAt) + seenAt time.Time // executor first handled the header + frontierAtSeen uint64 + declined string // eligibility reason, when the header was declined + discarded string // discard reason, when a stream opened and was thrown away + openedAt time.Time +} + +// streamingSlot is one in-progress stream. +type streamingSlot struct { + slot uint64 + generation turbine.StreamGeneration + parentSlot uint64 + parentID solana.Hash + exec *blockExecution + // origin holds the block's own transaction objects in executed order; + // the bank executes stream-owned copies (block.ExecutionCopies), and the + // handshake proves the block by these originals. + origin []*solana.Transaction + nextStart uint32 + pending map[uint32]*turbine.StreamBatch + footerSeen bool + completed bool + openedAt time.Time + headerAt time.Time + groups []streamingGroup + verificationWait metrics.Timing + // timeline: what bounded the open (see metrics.StreamingExecution). Kept + // here rather than in the collector because the loop resets the collector + // before every wait and the block may arrive several waits after the open. + headerSeenAt time.Time + parentFullNanos int64 + parentAdmittedAt time.Time + parentReplayedAt time.Time + waitEnteredAt time.Time + // restoreSysvarCache puts the legacy process-global sysvar cache back to + // its state before the bank opened; nil when nothing was published. + restoreSysvarCache func() +} + +// streamingExecutor is owned by the replay loop and driven from its select. +type streamingExecutor struct { + deps streamingDeps + current *streamingSlot + // headers remembers header batches for slots ahead of the frontier so the + // next slot can open as soon as its parent finishes, even when its header + // wake-up arrived earlier. + headers map[uint64]*turbine.StreamBatch + // retired is the last generation per slot that this executor discarded or + // declined; header recovery (recoverHeader) never reopens it. Pruned with + // headers. + retired map[uint64]turbine.StreamGeneration + // observed is the per-slot header timeline (see streamingObservation), + // pruned with headers. + observed map[uint64]*streamingObservation + ticker *time.Ticker + // Verification observation and group execution hooks; tests substitute them. + waitVerificationFn func(context.Context, *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) + executeFn func(exec *blockExecution, txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerifySignatures bool) error +} + +func newStreamingExecutor(deps streamingDeps) *streamingExecutor { + return &streamingExecutor{ + deps: deps, + headers: make(map[uint64]*turbine.StreamBatch), + retired: make(map[uint64]turbine.StreamGeneration), + observed: make(map[uint64]*streamingObservation), + executeFn: (*blockExecution).executeTransactionGroup, + waitVerificationFn: func(ctx context.Context, batch *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) { + return batch.WaitVerification(ctx) + }, + } +} + +// tick returns the polling channel, which is nil (never fires) while no +// stream is open, so the replay loop's select stays quiet when idle. +func (s *streamingExecutor) tick() <-chan time.Time { + if s == nil || s.current == nil { + return nil + } + if s.ticker == nil { + s.ticker = time.NewTicker(streamingPollInterval) + } + return s.ticker.C +} + +func (s *streamingExecutor) stopTicker() { + if s.ticker != nil { + s.ticker.Stop() + s.ticker = nil + } +} + +// events is the feed channel the wait loop selects on; nil when the feed is +// off, which never fires. +func (s *streamingExecutor) events() <-chan turbine.StreamEvent { + if s == nil || s.deps.feed == nil { + return nil + } + return s.deps.feed.StreamEvents() +} + +// matches reports whether a stream is open for slot. +func (s *streamingExecutor) matches(slot uint64) bool { + return s != nil && s.current != nil && s.current.slot == slot +} + +// beforeBlock restores speculative state before an intervening real bank is +// observed. A skip has no bank changes and must not throw away a later child. +func (s *streamingExecutor) beforeBlock(block *b.Block) { + if s != nil && s.current != nil && !block.IsSkipped && !s.matches(block.Slot) { + s.discard("other_block") + } +} + +// shutdown discards any open stream; the replay loop defers it so an exiting +// attempt never leaves a speculative bank (and its watchdog) behind. +func (s *streamingExecutor) shutdown() { + if s == nil { + return + } + s.discard("shutdown") + s.stopTicker() + s.headers = make(map[uint64]*turbine.StreamBatch) + s.retired = make(map[uint64]turbine.StreamGeneration) + s.observed = make(map[uint64]*streamingObservation) +} + +// handleEvent consumes one feed wake-up. +func (s *streamingExecutor) handleEvent(event turbine.StreamEvent) { + if s == nil { + return + } + var live bool + event, live = event.Resolve() + if !live { + return + } + switch event.Kind { + case turbine.StreamBatchReady: + if event.Batch == nil { + return + } + if s.current != nil && event.Generation == s.current.generation { + // A previous group can leave many notifications queued. Refresh + // the authoritative ready set before choosing the next group. + if event.Batch.Start < s.current.nextStart { + return // already consumed by a prior refresh + } + s.offer(event.Batch) + s.pull() + s.consume() + return + } + if event.Batch.Marker == turbine.StreamMarkerHeader && event.Batch.Start == 0 { + s.rememberHeader(event.Batch) + } else { + s.recoverHeader(event.Slot, event.Generation) + } + s.tryOpen() + case turbine.StreamCancelled: + if s.current != nil && event.Generation == s.current.generation { + s.discard("cancelled:" + event.Reason) + } + if header, ok := s.headers[event.Slot]; ok && header.Generation == event.Generation { + delete(s.headers, event.Slot) + } + s.tryOpen() + case turbine.StreamCompleted: + if s.current != nil && event.Generation == s.current.generation { + s.current.completed = true + // Released prefetch results remain immutable and owned by this + // generation. Recover ready batches whose wake-ups were dropped. + s.pull() + s.consume() + return + } + if header, ok := s.headers[event.Slot]; ok && header.Generation == event.Generation { + delete(s.headers, event.Slot) + } + } +} + +// handleTick polls the assembler for batches (recovery after dropped +// wake-ups), enforces the open-age bound, and opens the next slot if idle. +func (s *streamingExecutor) handleTick() { + if s == nil { + return + } + if s.current == nil { + s.tryOpen() + return + } + cur := s.current + switch s.deps.feed.StreamStatusOf(cur.generation) { + case turbine.StreamGone: + s.discard("gone") + s.tryOpen() + return + case turbine.StreamDone: + cur.completed = true + } + // An incomplete slot is bounded by MaxOpenAge (its shreds stopped + // arriving); a completed one may legitimately wait longer in the emitter + // (ancestry decisions) and is only bounded to cap the overlay's lifetime. + age := time.Since(cur.openedAt) + if (!cur.completed && age > StreamingExecutionCfg.maxOpenAge()) || age > streamingHardOpenAgeFactor*StreamingExecutionCfg.maxOpenAge() { + s.discard("timeout") + s.tryOpen() + return + } + s.pull() + s.consume() +} + +func (s *streamingExecutor) rememberHeader(header *turbine.StreamBatch) { + frontier := s.deps.frontier() + if header.Slot <= frontier || header.Slot-frontier > streamingMaxSlotDistance { + return + } + s.headers[header.Slot] = header + if obs := s.observed[header.Slot]; obs == nil || obs.generation != header.Generation { + s.observed[header.Slot] = &streamingObservation{ + generation: header.Generation, + parentSlot: header.ParentSlot, + readyAt: header.ReadyAt, + seenAt: time.Now(), + frontierAtSeen: frontier, + } + } + s.pruneHeaders(frontier) +} + +// recoverHeader handles a wake-up for a batch of a generation whose header +// this executor has not seen: the header's own wake-up may have been dropped +// (full channel), in which case the assembler is the authoritative source. +// Only slots within the bounded lookahead are worth the lookup, and a generation this +// executor already retired (discarded, or declined as ineligible) is never +// brought back: whole-block execution owns it from then on. A recovered +// header retains its original readiness time, including time before this lookup. +func (s *streamingExecutor) recoverHeader(slot uint64, g turbine.StreamGeneration) { + if g.IsZero() { + return + } + anchor := s.deps.frontier() + if s.current != nil { + anchor = s.current.slot + } + if slot <= anchor || slot-anchor > streamingMaxSlotDistance { + return + } + if known, ok := s.headers[slot]; ok && known.Generation == g { + return + } + if retired, ok := s.retired[slot]; ok && retired == g { + return + } + // Sorted by start: the header is the first batch, at 0, or not decoded. + pending := s.deps.feed.PendingStreamBatches(g, 0) + if len(pending) > 0 && pending[0].Start == 0 && pending[0].Marker == turbine.StreamMarkerHeader { + s.rememberHeader(pending[0]) + } +} + +// retire records that generation g of slot must not open again through +// header recovery; discard and the ineligible path call it. +func (s *streamingExecutor) retire(slot uint64, g turbine.StreamGeneration) { + if g.IsZero() { + return + } + s.retired[slot] = g +} + +// pruneHeaders bounds the header and retired maps: anything at or below the +// frontier can never open. +func (s *streamingExecutor) pruneHeaders(frontier uint64) { + for slot := range s.headers { + if slot <= frontier || slot-frontier > streamingMaxSlotDistance { + delete(s.headers, slot) + } + } + for slot := range s.retired { + if slot <= frontier || slot-frontier > streamingMaxSlotDistance { + delete(s.retired, slot) + } + } + for slot := range s.observed { + if slot <= frontier || slot-frontier > streamingMaxSlotDistance { + delete(s.observed, slot) + } + } +} + +// nextHeader prefers the next slot, otherwise the earliest nearby child of +// the executed bank. A header is only a speculation hint: it does not prove +// skips, advance the frontier, or authorize publication or voting. +func (s *streamingExecutor) nextHeader(frontier uint64) *turbine.StreamBatch { + last := s.deps.lastSlotCtx() + var selected *turbine.StreamBatch + for slot, header := range s.headers { + if slot <= frontier || slot-frontier > streamingMaxSlotDistance { + continue + } + if slot-frontier != 1 && (last == nil || header.ParentSlot != last.Slot) { + continue + } + if selected == nil || slot < selected.Slot { + selected = header + } + } + return selected +} + +// tryOpen may speculate across unresolved slots only on the exact executed +// parent. Real intervening blocks and fork switches discard the overlay; +// skip records leave it alone. The complete-block handshake stays mandatory. +func (s *streamingExecutor) tryOpen() { + if s == nil || s.current != nil || !StreamingExecutionCfg.Enabled { + return + } + frontier := s.deps.frontier() + s.pruneHeaders(frontier) + for { + header := s.nextHeader(frontier) + if header == nil { + return + } + delete(s.headers, header.Slot) + if reason := s.eligibility(header); reason != "" { + mlog.Log.FileOnlyf("streaming: slot %d not opened (%s)", header.Slot, reason) + s.retire(header.Slot, header.Generation) + if obs := s.observed[header.Slot]; obs != nil && obs.generation == header.Generation { + obs.declined = reason + } + continue + } + s.openStream(header) + return + } +} + +// eligibility returns an empty string when a stream may open on header, or +// the reason it may not. +func (s *streamingExecutor) eligibility(header *turbine.StreamBatch) string { + d := s.deps + if !d.alpenglowMode || !d.unrootedTailUsed || d.tail == nil { + return "requires alpenglow rooted-durable replay" + } + if d.feed.StreamStatusOf(header.Generation) != turbine.StreamActive { + return "generation no longer active" + } + last := d.lastSlotCtx() + if last == nil { + return "no executed parent context" + } + if frontier := d.frontier(); header.Slot <= frontier || header.Slot-frontier > streamingMaxSlotDistance || header.ParentSlot != last.Slot { + return fmt.Sprintf("slot %d on parent %d does not extend the executed frontier %d (parent context %d)", header.Slot, header.ParentSlot, frontier, last.Slot) + } + executedID, ok := d.executedBlockID(last.Slot) + if !ok || executedID == (solana.Hash{}) || executedID != header.ParentBlockID { + return "parent block id does not match the executed parent" + } + if d.switchPending() { + return "fork switch pending" + } + if d.epochSchedule == nil || d.epochSchedule.GetEpoch(header.Slot) != d.currentEpoch() { + return "epoch boundary" + } + if d.rewardsInFlight() { + return "partitioned rewards in flight" + } + if d.currentFeatures() == nil { + return "no feature set" + } + return "" +} + +func (s *streamingExecutor) openStream(header *turbine.StreamBatch) { + d := s.deps + last := d.lastSlotCtx() + shell := &b.Block{ + Slot: header.Slot, + SourceParentSlot: header.ParentSlot, + FromLiveStream: true, + AlpenglowParentBlockID: header.ParentBlockID, + HasAlpenglowParentBlockID: true, + } + shell.Epoch = d.epochSchedule.GetEpoch(shell.Slot) + // Same derivation the loop applies to the complete block, minus the + // process-global "current slot" publication, which stays at the frontier + // until the complete block is configured. + if err := configureBlockFromParent(shell, last, d.epochSchedule, false); err != nil { + mlog.Log.Warnf("streaming: slot %d not opened: %v", shell.Slot, err) + return + } + // The parent's VoteTimestamps map is shared by reference through + // configureBlock; a speculative bank must mutate its own copy. + shell.VoteTimestamps = maps.Clone(last.VoteTimestamps) + shell.Features = d.currentFeatures() + + // Bank open publishes the child's derived Clock/SlotHashes to the legacy + // process-global sysvar cache (Alpenglow banks never read it back — they + // pin from parentBankSysvars — but RPC simulation may). Snapshot it so a + // discard restores the parent's view; the accepted bank leaves it as a + // whole-block open would have. + sysvarCacheAtOpen := sealevel.SysvarCache + exec := newBlockExecution(d.acctsDb, shell, d.epochSchedule, StreamingExecutionCfg.workers(d.txParallelism), d.dbgOpts, d.persistedHashes, d.tail, d.transactionStatuses, d.alpenglowClock, last.BankSysvars()) + if err := exec.open(); err != nil { + exec.close() + sealevel.SysvarCache = sysvarCacheAtOpen + mlog.Log.Warnf("streaming: slot %d not opened: %v", shell.Slot, err) + return + } + exec.slotCtx.DeferVoteCachePublication = true + exec.slotCtx.TrackProgramCacheAdds = true + exec.setReplayStage("streaming_wait") + + cur := &streamingSlot{ + slot: shell.Slot, + generation: header.Generation, + parentSlot: header.ParentSlot, + parentID: header.ParentBlockID, + exec: exec, + pending: make(map[uint32]*turbine.StreamBatch), + openedAt: time.Now(), + headerAt: header.ReadyAt, + headerSeenAt: header.ReadyAt, + restoreSysvarCache: func() { sealevel.SysvarCache = sysvarCacheAtOpen }, + } + if obs := s.observed[shell.Slot]; obs != nil && obs.generation == header.Generation { + obs.openedAt = cur.openedAt + if !obs.seenAt.IsZero() { + cur.headerSeenAt = obs.seenAt + } + } + if d.frontierMark != nil { + // The mark is only the child's parent when the last executed block is + // that very slot; after a fork-switch re-base it is not, and the + // timeline says "unknown" rather than blaming the wrong slot. + if mark := d.frontierMark(); mark.slot == header.ParentSlot && !mark.replayedAt.IsZero() { + cur.parentFullNanos = mark.fullNanos + cur.parentAdmittedAt = mark.admittedAt + cur.parentReplayedAt = mark.replayedAt + cur.waitEnteredAt = mark.waitEnteredAt + } + } + s.current = cur + metrics.GlobalBlockReplay.StreamingExecution.Opened = 1 + d.feed.PrioritizeStreamRepair(header.Generation) + mlog.Log.FileOnlyf("streaming: opened slot %d on parent %d | %s", shell.Slot, header.ParentSlot, cur.openTimeline()) + s.offer(header) + s.pull() + s.consume() +} + +func (s *streamingExecutor) offer(batch *turbine.StreamBatch) { + cur := s.current + if cur == nil || batch == nil || batch.Start < cur.nextStart { + return + } + if _, seen := cur.pending[batch.Start]; !seen { + cur.pending[batch.Start] = batch + } +} + +// pull asks the assembler for everything decoded since nextStart; it is the +// authoritative path after a dropped wake-up. +func (s *streamingExecutor) pull() { + cur := s.current + if cur == nil { + return + } + for _, batch := range s.deps.feed.PendingStreamBatches(cur.generation, cur.nextStart) { + s.offer(batch) + } +} + +// consume executes every contiguous ready batch from nextStart as one group. +// Nothing is removed from pending until the group is committed, so holding +// for the group minimum leaves markers and batches exactly where they were. +func (s *streamingExecutor) consume() { + cur := s.current + if cur == nil { + return + } + var group []*turbine.StreamBatch + next := cur.nextStart + footer := false + for { + batch, ok := cur.pending[next] + if !ok { + break + } + if batch.Err != nil { + s.discard("decode_error") + return + } + switch batch.Marker { + case turbine.StreamMarkerHeader: + // The header opened the stream; nothing to execute. + case turbine.StreamMarkerUpdateParent: + // The leader abandoned the optimistic prefix we executed. + s.discard("update_parent") + return + case turbine.StreamMarkerFooter: + footer = true + default: + group = append(group, batch) + } + next = batch.End + 1 + } + if minBatches := StreamingExecutionCfg.MinGroupBatches; minBatches > 1 && len(group) > 0 && len(group) < minBatches && !cur.completed { + return // not enough ready work yet; everything stays pending + } + for start := cur.nextStart; start < next; { + batch := cur.pending[start] + delete(cur.pending, start) + start = batch.End + 1 + } + cur.nextStart = next + if footer { + cur.footerSeen = true + } + if len(group) == 0 { + return + } + if err := s.executeGroup(group); err != nil { + s.discard(err.Error()) + } +} + +// executeGroup joins verification for every batch in the group and executes +// the group's transactions as one unit. +func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { + cur := s.current + var txs []*solana.Transaction + var verified []txverify.VerifiedMessageIdentity + allVerified := true + readyAt := time.Now() + deadline := readyAt.Add(streamingVerificationWait) + maxAge := StreamingExecutionCfg.maxOpenAge() + if cur.completed { + maxAge *= streamingHardOpenAgeFactor + } + if ageDeadline := cur.openedAt.Add(maxAge); ageDeadline.Before(deadline) { + deadline = ageDeadline + } + ctx, cancel := context.WithDeadline(context.Background(), deadline) + defer cancel() + cur.exec.setReplayStage("streaming_sigverify_wait") + // Keep failure timings across replay-loop metric resets, just like headers. + joinStarted := time.Now() + recordJoin := func() { + cur.verificationWait.AddTiming(time.Since(joinStarted)) + if obs := s.observed[cur.slot]; obs != nil && obs.generation == cur.generation { + obs.verificationWait = cur.verificationWait + } + cur.exec.setReplayStage("streaming_wait") + } + for _, batch := range group { + identities, ok, err := s.waitVerificationFn(ctx, batch) + if errors.Is(err, turbine.ErrStreamBatchUnverified) { + ok, err = false, nil + } + if err != nil { + recordJoin() + if errors.Is(err, context.DeadlineExceeded) { + return errors.New("sigverify_timeout") + } + return fmt.Errorf("sigverify: %w", err) + } + if !ok { + allVerified = false + } + txs = append(txs, batch.Transactions...) + verified = append(verified, identities...) + } + recordJoin() + joinedAt := time.Now() + if len(txs) == 0 { + return nil + } + if !allVerified || len(verified) != len(txs) { + // The assembler's verifier refused the batch (admission), so the + // block-level verification at completion will cover it. Verifying here + // through ProcessTransaction is not an option: that path halts the + // process on an invalid signature, which a speculative bank on an + // unauthenticated prefix must never do. + return errors.New("unverified_batch") + } + // The identities are bound to the block's objects by the verifier; that + // binding is checked here, on the originals. The bank then executes + // copies made from those very originals (see block.ExecutionCopies): the + // block's objects are never resolved or otherwise mutated by a stream, + // so a discard leaves them exactly as turbine decoded them. + preparedForOriginals, err := b.PrepareVerifiedTransactionMessageIdentities(txs, verified) + if err != nil { + return fmt.Errorf("identities: %w", err) + } + copies, prepared, err := preparedForOriginals.ExecutionCopies() + if err != nil { + if errors.Is(err, b.ErrTransactionAlreadyResolved) { + return errors.New("resolved_input") + } + return fmt.Errorf("copies: %w", err) + } + started := time.Now() + err = s.executeFn(cur.exec, copies, prepared, false) + cur.exec.setReplayStage("streaming_wait") + if err != nil { + var duplicates *DuplicateTransactionMessagesError + if errors.As(err, &duplicates) { + return errors.New("duplicate_message") + } + if IsAlreadyProcessedTransactionError(err) { + return errors.New("already_processed") + } + return fmt.Errorf("group: %w", err) + } + cur.origin = append(cur.origin, txs...) + cur.groups = append(cur.groups, streamingGroup{readyAt: readyAt, joinedAt: joinedAt, startedAt: started, finishedAt: time.Now(), batches: len(group), transactions: len(txs)}) + return nil +} + +// discard throws the in-progress stream away and undoes every side effect it +// may have had outside its own SlotCtx. +func (s *streamingExecutor) discard(reason string) { + if s == nil || s.current == nil { + return + } + cur := s.current + s.current = nil + s.deps.feed.PrioritizeStreamRepair(turbine.StreamGeneration{}) + s.stopTicker() + s.retire(cur.slot, cur.generation) + if obs := s.observed[cur.slot]; obs != nil && obs.generation == cur.generation { + obs.discarded = reason + } + exec := cur.exec + if exec != nil { + exec.close() + if exec.slotCtx != nil { + exec.slotCtx.TrackProgramCacheAdds = false + for _, key := range exec.slotCtx.TakeProgramCacheAdds() { + if s.deps.acctsDb != nil { + s.deps.acctsDb.RemoveProgramFromCache(key) + } + } + // Deferred vote-cache changes die with the SlotCtx; the stake index + // entries belong to this slot; a concurrent leader bank may own + // entries at later slots. + exec.slotCtx.PendingVoteCache = nil + exec.slotCtx.PendingVoteCacheDeletes = nil + exec.slotCtx.VoteStakeDirty = false + } + } + global.DropPendingStakePubkeys(cur.slot) + if cur.restoreSysvarCache != nil { + cur.restoreSysvarCache() + } + metrics.GlobalBlockReplay.StreamingExecution.VerificationWait = cur.verificationWait + metrics.GlobalBlockReplay.StreamingExecution.Discarded = 1 + metrics.GlobalBlockReplay.StreamingExecution.DiscardReason = reason + mlog.Log.FileOnlyf("streaming: discarded slot %d after %d groups (%s)", cur.slot, len(cur.groups), reason) +} + +// discardSlot discards the stream if it is open for slot. +func (s *streamingExecutor) discardSlot(slot uint64, reason string) { + if s.matches(slot) { + s.discard(reason) + } +} + +// streamingFinalizeError marks a failure after the handshake passed; the +// block is as invalid as it would have been for the whole-block path. +type streamingFinalizeError struct{ err error } + +func (e *streamingFinalizeError) Error() string { return e.err.Error() } +func (e *streamingFinalizeError) Unwrap() error { return e.err } + +// finalize completes execution of block on the open stream. ok reports +// whether the stream matched the block; when it did not, the stream has been +// discarded and the caller must execute the block whole. A non-nil error with +// ok == true is a failure after the handshake and is final for the block, +// exactly as a ProcessBlock error is. +// +// Ownership: the stream keeps owning its bank (s.current) until the tail has +// committed, so every failure path after the handshake goes through the same +// discard as a pre-handshake mismatch — program-cache insertions evicted, +// unpublished vote-cache entries dropped, slot-keyed stake entries dropped, +// the legacy sysvar cache restored, the execution closed. The one publication +// that precedes the tail is the deferred vote cache, applied at the point +// where whole-block execution would already have written it (before fees, +// rent, footer and bank hash); a failure inside the tail therefore leaves the +// same footprint a whole-block tail failure leaves, and the dirty marker it +// sets is what forces the rooted-checkpoint re-replay on recovery. +func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel.BankSysvars) (slotCtx *sealevel.SlotCtx, ok bool, err error) { + if s == nil || s.current == nil || block == nil { + return nil, false, nil + } + cur := s.current + finalizeStart := time.Now() + if reason := s.handshake(block, parentBankSysvars); reason != "" { + s.discard("prefix_mismatch:" + reason) + return nil, false, nil + } + exec := cur.exec + fullAt := time.Time{} + if block.ShredFullNanos > 0 { + fullAt = time.Unix(0, block.ShredFullNanos) + } + + // Whole-block plan and status validation, exactly as ProcessBlock does + // them, now that the authoritative block exists. A failure here is not yet + // a verdict on the block: the whole-block path re-derives it. + if err := validateBlockTransactionVersions(block); err != nil { + s.discard("versions") + return nil, false, nil + } + executionPlanStart := time.Now() + executionPlan, err := planBlockTransactionExecution(block) + metrics.GlobalBlockReplay.TransactionExecutionPlan.AddTimingSince(executionPlanStart) + if err != nil { + s.discard("plan") + return nil, false, nil + } + executed := len(cur.origin) + if len(exec.transactions) != executed { + s.discard("prefix_bookkeeping") + return nil, false, nil + } + for i := 0; i < executed; i++ { + if executionPlan.execute[i] != exec.execute[i] { + s.discard("execution_mask") + return nil, false, nil + } + } + statusValidationStart := time.Now() + statusValidation, statusValidationErr := s.deps.transactionStatuses.validateBlockForPublication(block, executionPlan) + metrics.GlobalBlockReplay.TransactionStatusValidation.AddTimingSince(statusValidationStart) + if statusValidationErr != nil { + s.discard("status_validation") + return nil, false, nil + } + statusPreparation := s.deps.transactionStatuses.startStatusPreparation(executionPlan) + defer func() { + statusPreparation.wait() + if statusPreparation != nil { + metrics.GlobalBlockReplay.TransactionStatusPreparation.AddTiming(statusPreparation.duration) + } + }() + + // From here on the stream is committed to this block: any failure is the + // block's failure. fail undoes the stream's side effects and reports it. + s.stopTicker() + fail := func(reason string, err error) (*sealevel.SlotCtx, bool, error) { + s.discard("finalize:" + reason) + return nil, true, &streamingFinalizeError{err: err} + } + block.FeeRateGovernor = exec.block.FeeRateGovernor + block.VoteTimestamps = exec.slotCtx.VoteTimestamps + exec.block = block + exec.slotCtx.Blockhash = block.Blockhash + exec.slotCtx.Epoch = block.Epoch + if requireAlpenglowBlockFooter(block, exec.slotCtx, s.deps.alpenglowClock) { + if err := validateAlpenglowFooterNanosecondClock(exec.slotCtx, block); err != nil { + return fail("footer_clock", err) + } + } + if suffix := block.Transactions[executed:]; len(suffix) > 0 { + started := time.Now() + err := s.executeFn(exec, suffix, executionPlan.messageIdentities.Slice(executed, len(block.Transactions)), !block.TransactionSignaturesVerified()) + if err != nil { + return fail("suffix", fmt.Errorf("execute block suffix at slot %d: %w", block.Slot, err)) + } + cur.groups = append(cur.groups, streamingGroup{readyAt: started, joinedAt: started, startedAt: started, finishedAt: time.Now(), transactions: len(suffix), suffix: true}) + } + if exec.processedSignatures != executionPlan.processedSignatures || exec.processedTxCount != executionPlan.processedTxCount { + return fail("counts", fmt.Errorf("streaming execution at slot %d processed %d transactions/%d signatures, block plan has %d/%d", + block.Slot, exec.processedTxCount, exec.processedSignatures, executionPlan.processedTxCount, executionPlan.processedSignatures)) + } + exec.slotCtx.NumSignatures = executionPlan.processedSignatures + + // Acceptance of the executed transactions: publish what execution would + // have published as it ran, then run the unchanged tail. Program-cache + // insertions stay tracked until the tail commits so a tail failure can + // still evict them. + publishDeferredVoteCache(exec.slotCtx) + exec.executionPlan = executionPlan + exec.statusPreparation = statusPreparation + exec.statusValidation = statusValidation + slotCtx, err = exec.finalize() + if err != nil { + return fail("tail", err) + } + exec.slotCtx.TrackProgramCacheAdds = false + exec.slotCtx.TakeProgramCacheAdds() + s.current = nil + s.deps.feed.PrioritizeStreamRepair(turbine.StreamGeneration{}) + exec.close() + + // The per-block record is rebuilt from the stream's own bookkeeping: the + // loop resets the collector between waits, so counters accumulated while + // executing groups may or may not have survived to this point. + record := &metrics.GlobalBlockReplay.StreamingExecution + discarded, discardReason := record.Discarded, record.DiscardReason + *record = metrics.StreamingExecution{Opened: 1, Discarded: discarded, DiscardReason: discardReason} + record.VerificationWait = cur.verificationWait + record.Groups = uint64(len(cur.groups)) + for _, group := range cur.groups { + record.Transactions += uint64(group.transactions) + // Work that finished before the last shred arrived is the latency the + // stream took off the vote path. + if !fullAt.IsZero() && group.finishedAt.Before(fullAt) { + record.TxLoopBeforeFull.AddTiming(group.finishedAt.Sub(group.startedAt)) + } + } + record.OpenDelay.AddTiming(cur.openedAt.Sub(cur.headerAt)) + cur.recordTimeline(record, block, finalizeStart) + cur.recordGroups(record, fullAt) + // Publish only the accepted stream, including its open, early groups and + // suffix. The finalization record already contains any tail loader work. + metrics.GlobalBlockReplay.AccountLoader.Accumulate(exec.accountLoader) + return slotCtx, true, nil +} + +// handshake proves the executed prefix is the block. It returns the mismatch +// reason, or "" when every binding holds. +func (s *streamingExecutor) handshake(block *b.Block, parentBankSysvars *sealevel.BankSysvars) string { + cur := s.current + exec := cur.exec + switch { + case block.Slot != cur.slot: + return "slot" + case block.IsSkipped || !block.FromLiveStream: + return "not a live block" + case !block.HasAlpenglowParentBlockID || block.AlpenglowParentBlockID != cur.parentID || block.SourceParentSlot != cur.parentSlot: + return "parent" + case block.ParentSlot != exec.block.ParentSlot || block.ParentBankhash != exec.block.ParentBankhash: + return "configured parent" + case block.Features != exec.block.Features: + return "features" + case parentBankSysvars == nil || parentBankSysvars != exec.parentBankSysvars: + return "parent sysvars" + case block.Epoch != exec.block.Epoch: + return "epoch" + case len(block.EpochUpdatedAccts) != 0: + return "epoch account updates" + case len(block.Transactions) < len(cur.origin): + return "shorter than executed prefix" + } + status := s.deps.feed.StreamStatusOf(cur.generation) + if status == turbine.StreamGone { + return "generation gone" + } + for i, tx := range cur.origin { + if block.Transactions[i] != tx { + return fmt.Sprintf("transaction %d identity", i) + } + } + return "" +} + +// nanosOf is a time as unix nanoseconds, 0 for the zero time. +func nanosOf(t time.Time) int64 { + if t.IsZero() { + return 0 + } + return t.UnixNano() +} + +// recordTimeline writes the stream's open timeline and its decomposition +// into the block's record. Every wait is attributed to exactly one thing: +// +// header ready ─(parent arrival)─▶ parent full ─(parent replay)─▶ parent +// replayed ─(loop: post-replay tail │ dispatch)─▶ opened +// +// with the header's own wake-up latency (ready → seen) reported alongside. +// Each component is zero when the timeline cannot support it (unknown +// instants, or an ordering that makes it empty). The arithmetic is done on +// the wall-clock nanos the record carries, so the components and the +// instants are exactly consistent for whoever joins them later. +func (cur *streamingSlot) recordTimeline(record *metrics.StreamingExecution, block *b.Block, finalizeStart time.Time) { + ready, seen, opened := nanosOf(cur.headerAt), nanosOf(cur.headerSeenAt), nanosOf(cur.openedAt) + if seen == 0 { + seen = ready + } + parentFull, parentAdmitted, parentReplayed, waitEntered := cur.parentFullNanos, nanosOf(cur.parentAdmittedAt), nanosOf(cur.parentReplayedAt), nanosOf(cur.waitEnteredAt) + record.HeaderReadyNanos = ready + record.HeaderSeenNanos = seen + record.OpenedNanos = opened + record.ParentFullNanos = parentFull + record.ParentAdmittedNanos = parentAdmitted + record.ParentReplayedNanos = parentReplayed + record.WaitEnteredNanos = waitEntered + if len(cur.groups) > 0 { + record.FirstGroupStartNanos = nanosOf(cur.groups[0].startedAt) + } + if block != nil && block.ShredFullNanos > 0 { + record.FullNanos = block.ShredFullNanos + } + record.FinalizeStartNanos = nanosOf(finalizeStart) + + add := func(timing *metrics.Timing, from, to int64) { + if from > 0 && to > from { + timing.AddTiming(time.Duration(to - from)) + } + } + add(&record.OpenWaitParentArrival, ready, parentFull) + if parentReplayed > 0 { + replayStart := max(ready, parentFull) + add(&record.OpenWaitParentReplay, replayStart, parentReplayed) + // Admission is a milestone, not an execution boundary: streaming + // can execute inside waitForReplayInput before admission. + if parentAdmitted > 0 { + add(&record.OpenWaitParentPreAdmission, replayStart, min(parentAdmitted, parentReplayed)) + add(&record.OpenWaitParentPostAdmission, max(parentAdmitted, replayStart), parentReplayed) + } + } + loopStart := max(ready, parentReplayed) + add(&record.OpenWaitLoop, loopStart, opened) + // The loop's wait splits at its first entry into the replay wait after + // the parent: before it is the parent's post-replay tail, after it the + // dispatch of the child's header (queued events ahead of it, the poll). + if waitEntered > 0 && parentReplayed > 0 && waitEntered > parentReplayed { + add(&record.OpenWaitPostReplay, loopStart, min(waitEntered, opened)) + add(&record.OpenWaitDispatch, max(waitEntered, loopStart), opened) + } +} + +// recordGroups writes the per-group view: joining/assembly, preparation, +// and execution elapsed intervals, including how much of each interval +// fell after the last shred (the part FullToReplayed pays for), and the +// largest group (a late open turns the whole backlog into one group). The +// suffix counts as a group that starts after full. +func (cur *streamingSlot) recordGroups(record *metrics.StreamingExecution, fullAt time.Time) { + // Without a full instant nothing is "after full" (as TxLoopBeforeFull + // and FullToReplayed record nothing either); waits and sizes still count. + after := func(from, to time.Time) time.Duration { + if fullAt.IsZero() { + return 0 + } + if fullAt.After(from) { + from = fullAt + } + if to.After(from) { + return to.Sub(from) + } + return 0 + } + for _, group := range cur.groups { + if wait := group.joinedAt.Sub(group.readyAt); wait > 0 && !group.suffix { + record.GroupJoinAssembly.AddTiming(wait) + if late := after(group.readyAt, group.joinedAt); late > 0 { + record.GroupJoinAssemblyAfterFull.AddTiming(late) + } + } + if !group.suffix && group.startedAt.After(group.joinedAt) { + record.GroupPreparation.AddTiming(group.startedAt.Sub(group.joinedAt)) + if late := after(group.joinedAt, group.startedAt); late > 0 { + record.GroupPreparationAfterFull.AddTiming(late) + } + } + if late := after(group.startedAt, group.finishedAt); late > 0 { + record.TxLoopAfterFull.AddTiming(late) + if !group.suffix && !fullAt.IsZero() && group.startedAt.Before(fullAt) { + record.GroupsStraddlingFull++ + } + } + if uint64(group.transactions) > record.LargestGroupTransactions { + record.LargestGroupTransactions = uint64(group.transactions) + record.LargestGroupBatches = uint64(group.batches) + } + } + if n := len(cur.groups); n > 0 { + record.LastGroupEndNanos = nanosOf(cur.groups[n-1].finishedAt) + } + // The tail cases are worth a per-group line; bounded so a heavy block + // with hundreds of groups does not flood the log. + if record.TxLoopAfterFull.SumNanoseconds > uint64(30*time.Millisecond) || record.GroupJoinAssemblyAfterFull.SumNanoseconds > uint64(5*time.Millisecond) { + mlog.Log.FileOnlyf("streaming: slot %d groups (vs last shred): %s", cur.slot, cur.groupTimeline(fullAt, 12)) + } +} + +// groupTimeline renders up to limit groups (the first ones and the last) +// relative to fullAt: "[#0 b=3 tx=1200 ready-150.2 joined-149.8 exec-149.8..-140.1]". +func (cur *streamingSlot) groupTimeline(fullAt time.Time, limit int) string { + rel := func(t time.Time) string { + if t.IsZero() || fullAt.IsZero() { + return "?" + } + return fmt.Sprintf("%+.1f", float64(t.Sub(fullAt).Microseconds())/1e3) + } + var out []byte + render := func(i int) { + g := cur.groups[i] + kind := "" + if g.suffix { + kind = " suffix" + } + out = fmt.Appendf(out, "[#%d%s b=%d tx=%d ready%s joined%s exec%s..%s]", i, kind, g.batches, g.transactions, rel(g.readyAt), rel(g.joinedAt), rel(g.startedAt), rel(g.finishedAt)) + } + n := len(cur.groups) + if n <= limit { + for i := range cur.groups { + render(i) + } + return string(out) + } + for i := 0; i < limit-1; i++ { + render(i) + } + out = fmt.Appendf(out, "…(%d more)", n-limit) + render(n - 1) + return string(out) +} + +// openTimeline renders the open's timeline for the log, relative to the +// header's decode instant. +func (cur *streamingSlot) openTimeline() string { + base := nanosOf(cur.headerAt) + rel := func(nanos int64) string { + if nanos == 0 { + return "?" + } + return fmt.Sprintf("%+.1fms", float64(nanos-base)/1e6) + } + return fmt.Sprintf("header seen %s, parent full %s, parent admitted %s, parent replayed %s, wait entered %s, opened %s (vs header ready)", + rel(nanosOf(cur.headerSeenAt)), rel(cur.parentFullNanos), rel(nanosOf(cur.parentAdmittedAt)), rel(nanosOf(cur.parentReplayedAt)), rel(nanosOf(cur.waitEnteredAt)), rel(nanosOf(cur.openedAt))) +} + +// noteWholeBlock records, for a block about to execute whole, why no stream +// opened for it (the block's record otherwise only says Opened == 0). The +// header timeline is filled in when the header was seen, so the analysis can +// tell "never decoded a header" from "decoded one and could not use it". +func (s *streamingExecutor) noteWholeBlock(block *b.Block) { + if s == nil || block == nil || block.IsSkipped { + return + } + record := &metrics.GlobalBlockReplay.StreamingExecution + if block.ShredFullNanos > 0 { + record.FullNanos = block.ShredFullNanos + } + obs := s.observed[block.Slot] + switch { + case obs == nil: + record.NotOpenedReason = "header_not_seen" + return + case obs.discarded != "": + record.NotOpenedReason = "discarded:" + obs.discarded + case obs.declined != "": + record.NotOpenedReason = "declined:" + obs.declined + case !obs.openedAt.IsZero(): + // Unreachable in practice (an opened stream ends in finalize or in a + // discard, which records itself); kept so the record never lies. + record.NotOpenedReason = "opened_not_discarded" + default: + record.NotOpenedReason = fmt.Sprintf("waiting_for_parent:header_on_parent_%d_seen_at_frontier_%d", obs.parentSlot, obs.frontierAtSeen) + } + record.VerificationWait = obs.verificationWait + record.HeaderReadyNanos = nanosOf(obs.readyAt) + record.HeaderSeenNanos = nanosOf(obs.seenAt) + record.OpenedNanos = nanosOf(obs.openedAt) +} diff --git a/pkg/replay/streaming_lifecycle_test.go b/pkg/replay/streaming_lifecycle_test.go new file mode 100644 index 000000000..6ee885ef8 --- /dev/null +++ b/pkg/replay/streaming_lifecycle_test.go @@ -0,0 +1,593 @@ +package replay + +import ( + "bytes" + "context" + "encoding/binary" + "sort" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/arena" + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/Overclock-Validator/mithril/pkg/txverify" + bin "github.com/gagliardetto/binary" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// The lifecycle equivalence gate: the same block executed whole by +// ProcessBlock and executed as a stream (bank opened on a transaction-less +// shell, groups fed through the executor, finalize against the complete +// block) must produce the same bank hash, the same committed account delta, +// the same signature/fee/compute totals and the same status publication. +// The bank is self-contained: an in-memory tail stands in for the unrooted +// working set, the parent bank sysvars are an explicit snapshot, rent +// rewrites are skipped (the rent scan needs a real AccountsDB), and the +// AccountsDB is only referenced for its directory. + +// lifecycleTail is an unrooted working set over a fixed durable memory: reads +// resolve to clones (or a zero-lamport placeholder), commits are recorded. +type lifecycleTail struct { + unrootedState + durable accounts.MemAccounts + added []lifecycleCommit +} + +type lifecycleCommit struct { + slot uint64 + delta []*accounts.Account + bankhash []byte +} + +func (t *lifecycleTail) GetAccount(_ uint64, pubkey solana.PublicKey) (*accounts.Account, error) { + if acct, err := t.durable.GetAccountWithoutLock(pubkey); err == nil { + return acct.Clone(), nil + } + return &accounts.Account{Key: pubkey}, nil +} + +func (t *lifecycleTail) GetAccountsBatch(_ context.Context, slot uint64, pks []solana.PublicKey) ([]*accounts.Account, error) { + out := make([]*accounts.Account, len(pks)) + for i, pk := range pks { + out[i], _ = t.GetAccount(slot, pk) + } + return out, nil +} + +func (t *lifecycleTail) Add(slot uint64, delta []*accounts.Account, bankhash []byte) { + t.added = append(t.added, lifecycleCommit{slot: slot, delta: delta, bankhash: append([]byte(nil), bankhash...)}) +} + +func (t *lifecycleTail) OverCap() bool { return false } + +type lifecycleEnv struct { + feats *features.Features + durable accounts.MemAccounts + acctsDb *accountsdb.AccountsDb + epochSchedule *sealevel.SysvarEpochSchedule + parent *sealevel.BankSysvars +} + +const ( + lifecycleParentSlot = uint64(7) + lifecycleSlot = uint64(8) +) + +var lifecycleParentBlockID = solana.Hash{0xAA, 0xBB} + +// ensureBorrowedAccountArenas gives parallelTxLoop the per-worker arena slots +// the node installs at startup (nil arenas are accepted by ProcessTransaction). +func ensureBorrowedAccountArenas(t *testing.T, n int) { + t.Helper() + if len(sealevel.BorrowedAccountArenas) >= n { + return + } + prev := sealevel.BorrowedAccountArenas + sealevel.BorrowedAccountArenas = make([]*arena.Arena[sealevel.BorrowedAccount], n) + t.Cleanup(func() { sealevel.BorrowedAccountArenas = prev }) +} + +func newLifecycleEnv(t *testing.T) *lifecycleEnv { + t.Helper() + ensureBorrowedAccountArenas(t, 4) + // Bank open publishes derived sysvars to the legacy process-global cache; + // leave it as we found it for the rest of the package. + sysvarCacheBefore := sealevel.SysvarCache + t.Cleanup(func() { sealevel.SysvarCache = sysvarCacheBefore }) + feats := features.NewFeaturesDefault() + feats.EnableFeature(features.FormalizeLoadedTransactionDataSize, 0) + feats.EnableFeature(features.SkipRentRewrites, 0) + + durable := accounts.NewMemAccounts() + _ = durable.SetAccountWithoutLock(addresses.SystemProgramAddr, &accounts.Account{ + Key: addresses.SystemProgramAddr, Lamports: 1, Owner: addresses.NativeLoaderAddr, Executable: true, RentEpoch: ^uint64(0), + }) + _ = durable.SetAccountWithoutLock(txfixture.PayerPubkey(), &accounts.Account{ + Key: txfixture.PayerPubkey(), Lamports: 3_200_000, Owner: addresses.SystemProgramAddr, RentEpoch: ^uint64(0), + }) + _ = durable.SetAccountWithoutLock(txfixture.DestPubkey(), &accounts.Account{ + Key: txfixture.DestPubkey(), Lamports: 10_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: ^uint64(0), + }) + + // The parent bank's sysvar snapshot, as the retained parent context would + // hold it. RecentBlockhashes carries the fixture blockhash so the transfers + // are age-valid. + clock := sealevel.SysvarClock{Slot: lifecycleParentSlot, EpochStartTimestamp: 111, UnixTimestamp: 222} + slotHashes := sealevel.SysvarSlotHashes{{Slot: lifecycleParentSlot - 1, Hash: [32]byte{0x61}}} + recent := sealevel.SysvarRecentBlockhashes{{ + Blockhash: txfixture.TestBlockhash(), + FeeCalculator: sealevel.FeeCalculator{LamportsPerSignature: 5_000}, + }} + slotHistory := sealevel.SysvarSlotHistory{ + Bits: sealevel.SlotHistoryBitvec{ + Bits: sealevel.SlotHistoryInner{BlocksLen: 1, Blocks: []uint64{0x81}}, + Len: 64, + }, + NextSlot: lifecycleSlot, + } + stakeHistory := sealevel.SysvarStakeHistory{{Epoch: 0, Entry: sealevel.StakeHistoryEntry{Effective: 91}}} + lastRestart := sealevel.SysvarLastRestartSlot{LastRestartSlot: 3} + epochSchedule := sealevel.SysvarEpochSchedule{SlotsPerEpoch: 100, LeaderScheduleSlotOffset: 100} + rent := sealevel.NewDefaultRentSysvar() + parent, err := sealevel.NewBankSysvars(lifecycleParentSlot, + &accounts.Account{Key: sealevel.SysvarClockAddr, Lamports: 1, Data: clock.MustMarshal()}, + &accounts.Account{Key: sealevel.SysvarSlotHashesAddr, Lamports: 1, Data: slotHashes.MustMarshal()}, + &accounts.Account{Key: sealevel.SysvarRecentBlockHashesAddr, Lamports: 1, Data: recent.MustMarshal()}, + &accounts.Account{Key: sealevel.SysvarSlotHistoryAddr, Lamports: 1, Data: slotHistory.MustMarshal()}, + &accounts.Account{Key: sealevel.SysvarStakeHistoryAddr, Lamports: 1, Data: marshalStakeHistoryForParentLoader(t, &stakeHistory)}, + &accounts.Account{Key: sealevel.SysvarLastRestartSlotAddr, Lamports: 1, Data: marshalLastRestartSlotForParentLoader(t, lastRestart)}, + &accounts.Account{Key: sealevel.SysvarEpochScheduleAddr, Lamports: 1, Data: marshalEpochScheduleForParentLoader(t, epochSchedule)}, + &accounts.Account{Key: sealevel.SysvarRentAddr, Lamports: 1, Data: rent.MustMarshal()}, + ) + require.NoError(t, err) + require.NoError(t, parent.ValidateForExecution()) + + return &lifecycleEnv{ + feats: feats, + durable: durable, + acctsDb: &accountsdb.AccountsDb{AcctsDir: t.TempDir()}, + epochSchedule: &epochSchedule, + parent: parent, + } +} + +// block builds the slot's block as the loop would have configured it on the +// executed parent; txs nil is the streaming shell. +func (env *lifecycleEnv) block(txs []*solana.Transaction) *b.Block { + return &b.Block{ + Slot: lifecycleSlot, + VoteTimestamps: make(map[solana.PublicKey]sealevel.BlockTimestamp), + Epoch: 0, + ParentSlot: lifecycleParentSlot, + ParentBankhash: [32]byte{0x88}, + Blockhash: [32]byte{0x99}, + LastBlockhash: [32]byte{0x77}, + Features: env.feats, + Transactions: txs, + PrevFeeRateGovernor: &sealevel.FeeRateGovernor{TargetLamportsPerSignature: 5_000, LamportsPerSignature: 5_000}, + FromLiveStream: true, + SourceParentSlot: lifecycleParentSlot, + AlpenglowParentBlockID: lifecycleParentBlockID, + HasAlpenglowParentBlockID: true, + } +} + +type lifecycleOutcome struct { + bankhash []byte + numSignatures uint64 + computeUnits uint64 + lamportsBurnt uint64 + delta map[solana.PublicKey]*accounts.Account +} + +func lifecycleOutcomeOf(t *testing.T, slotCtx *sealevel.SlotCtx, tail *lifecycleTail) lifecycleOutcome { + t.Helper() + require.NotNil(t, slotCtx) + require.Len(t, tail.added, 1, "the bank commits exactly once") + require.Equal(t, slotCtx.Slot, tail.added[0].slot) + require.Equal(t, slotCtx.FinalBankhash, tail.added[0].bankhash) + delta := make(map[solana.PublicKey]*accounts.Account, len(tail.added[0].delta)) + for _, acct := range tail.added[0].delta { + delta[acct.Key] = acct + } + require.Contains(t, delta, txfixture.PayerPubkey()) + return lifecycleOutcome{ + bankhash: append([]byte(nil), slotCtx.FinalBankhash...), + numSignatures: slotCtx.NumSignatures, + computeUnits: slotCtx.TotalComputeUnitsConsumed, + lamportsBurnt: slotCtx.LamportsBurnt, + delta: delta, + } +} + +func requireSameLifecycleOutcome(t *testing.T, want, got lifecycleOutcome) { + t.Helper() + require.NotEmpty(t, want.bankhash) + require.Equal(t, want.bankhash, got.bankhash, "bank hash") + require.Equal(t, want.numSignatures, got.numSignatures) + require.Equal(t, want.computeUnits, got.computeUnits) + require.Equal(t, want.lamportsBurnt, got.lamportsBurnt) + wantKeys := make([]string, 0, len(want.delta)) + for key := range want.delta { + wantKeys = append(wantKeys, key.String()) + } + gotKeys := make([]string, 0, len(got.delta)) + for key := range got.delta { + gotKeys = append(gotKeys, key.String()) + } + sort.Strings(wantKeys) + sort.Strings(gotKeys) + require.Equal(t, wantKeys, gotKeys, "committed account set") + for key, acct := range want.delta { + other := got.delta[key] + require.Equal(t, acct.Lamports, other.Lamports, "%s lamports", key) + require.Equal(t, acct.Owner, other.Owner, "%s owner", key) + require.Equal(t, acct.Data, other.Data, "%s data", key) + require.Equal(t, acct.Executable, other.Executable, "%s executable", key) + } +} + +func lifecycleWholeBlock(t *testing.T, env *lifecycleEnv, txs []*solana.Transaction, txParallelism int) lifecycleOutcome { + t.Helper() + tail := &lifecycleTail{durable: env.durable} + statuses := NewTransactionStatusCache() + block := env.block(txs) + block.MarkTransactionSignaturesVerified() + slotCtx, err := ProcessBlock(env.acctsDb, block, env.epochSchedule, txParallelism, nil, &persistedTracker{}, tail, statuses, false, env.parent) + require.NoError(t, err) + return lifecycleOutcomeOf(t, slotCtx, tail) +} + +// lifecycleStream opens the bank on a transaction-less shell, feeds txs in +// the given batch splits through the executor, and finalizes against the +// complete block; suffixTxs of the transactions are never streamed and +// execute at finalize. +func lifecycleStream(t *testing.T, env *lifecycleEnv, txs []*solana.Transaction, splits []int, suffixTxs int, workers int) (lifecycleOutcome, *streamingExecutor) { + t.Helper() + tail := &lifecycleTail{durable: env.durable} + statuses := NewTransactionStatusCache() + shell := env.block(nil) + exec := newBlockExecution(env.acctsDb, shell, env.epochSchedule, workers, nil, &persistedTracker{}, tail, statuses, false, env.parent) + require.NoError(t, exec.open()) + exec.slotCtx.DeferVoteCachePublication = true + exec.slotCtx.TrackProgramCacheAdds = true + + feed := newFakeStreamFeed() + gen := turbine.NewDetachedStreamGeneration(lifecycleSlot) + feed.status[gen] = turbine.StreamActive + s := newStreamingExecutor(streamingDeps{ + feed: feed, + epochSchedule: env.epochSchedule, + tail: tail, + transactionStatuses: statuses, + alpenglowMode: true, + unrootedTailUsed: true, + frontier: func() uint64 { return lifecycleParentSlot }, + lastSlotCtx: func() *sealevel.SlotCtx { return nil }, + currentFeatures: func() *features.Features { return env.feats }, + currentEpoch: func() uint64 { return 0 }, + rewardsInFlight: func() bool { return false }, + switchPending: func() bool { return false }, + executedBlockID: func(uint64) (solana.Hash, bool) { return lifecycleParentBlockID, true }, + }) + s.current = &streamingSlot{ + slot: lifecycleSlot, generation: gen, parentSlot: lifecycleParentSlot, parentID: lifecycleParentBlockID, + exec: exec, pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: time.Now(), + } + header := turbine.NewDetachedStreamMarker(gen, 0, 0, turbine.StreamMarkerHeader, lifecycleParentSlot, lifecycleParentBlockID) + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, Batch: header}) + + streamed := txs[:len(txs)-suffixTxs] + start, next := 0, uint32(1) + for _, end := range append(append([]int(nil), splits...), len(streamed)) { + if end <= start { + continue + } + group := streamed[start:end] + batch := turbine.NewDetachedStreamBatch(gen, next, next+uint32(len(group))-1, group, verifiedIdentities(t, group)) + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, Batch: batch}) + next += uint32(len(group)) + start = end + } + require.NotNil(t, s.current, "no group may have discarded the stream (%s)", metrics.GlobalBlockReplay.StreamingExecution.DiscardReason) + sameTransactions(t, streamed, s.current.origin) + sameCopies(t, streamed, exec.transactions) + + block := env.block(txs) + block.MarkTransactionSignaturesVerified() + slotCtx, ok, err := s.finalize(block, env.parent) + require.NoError(t, err) + require.True(t, ok, "the stream must accept its own block") + require.Nil(t, s.current) + require.True(t, exec.closed) + require.Equal(t, uint64(1), metrics.GlobalBlockReplay.StreamingExecution.Opened) + require.Equal(t, uint64(len(txs)), metrics.GlobalBlockReplay.StreamingExecution.Transactions) + return lifecycleOutcomeOf(t, slotCtx, tail), s +} + +func TestStreamingLifecycleMatchesWholeBlock(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + // Transfers of 999,001+ lamports at a 5,000-lamport fee from a + // 3,200,000-lamport payer: the first two succeed, the rest fail for rent + // and still pay, so every group boundary is a write dependency on the + // payer and the outcome is order-dependent. + txs := transferTransactions(t, 12, 999_000) + txCountBefore := global.TransactionCount() + + whole := lifecycleWholeBlock(t, newLifecycleEnv(t), txs, 0) + require.Equal(t, txCountBefore+uint64(len(txs)), global.TransactionCount()) + wholeParallel := lifecycleWholeBlock(t, newLifecycleEnv(t), txs, 4) + requireSameLifecycleOutcome(t, whole, wholeParallel) + + cases := []struct { + name string + splits []int + suffixTxs int + workers int + }{ + {"one group, no suffix", nil, 0, 1}, + {"three groups, no suffix", []int{3, 7}, 0, 4}, + {"per-transaction groups", []int{1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11}, 0, 2}, + {"two groups and a suffix", []int{4}, 3, 4}, + {"everything in the suffix", nil, 12, 4}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + before := global.TransactionCount() + streamed, _ := lifecycleStream(t, newLifecycleEnv(t), txs, tc.splits, tc.suffixTxs, tc.workers) + requireSameLifecycleOutcome(t, whole, streamed) + require.Equal(t, before+uint64(len(txs)), global.TransactionCount()) + }) + } +} + +// A stream discarded after executing groups leaves the durable view and the +// tail untouched, and the same block then executes whole to the same result. +func TestStreamingLifecycleDiscardThenWholeBlock(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + for _, reason := range []string{"update_parent", "sigverify_timeout"} { + t.Run(reason, func(t *testing.T) { + txs := transferTransactions(t, 8, 999_000) + env := newLifecycleEnv(t) + whole := lifecycleWholeBlock(t, env, txs, 2) + + tail := &lifecycleTail{durable: env.durable} + statuses := NewTransactionStatusCache() + shell := env.block(nil) + exec := newBlockExecution(env.acctsDb, shell, env.epochSchedule, 2, nil, &persistedTracker{}, tail, statuses, false, env.parent) + require.NoError(t, exec.open()) + feed := newFakeStreamFeed() + gen := turbine.NewDetachedStreamGeneration(lifecycleSlot) + feed.status[gen] = turbine.StreamActive + s := newStreamingExecutor(streamingDeps{feed: feed, epochSchedule: env.epochSchedule, tail: tail, transactionStatuses: statuses}) + s.current = &streamingSlot{slot: lifecycleSlot, generation: gen, parentSlot: lifecycleParentSlot, parentID: lifecycleParentBlockID, + exec: exec, pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: time.Now(), nextStart: 1} + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, + Batch: turbine.NewDetachedStreamBatch(gen, 1, 5, txs[:5], verifiedIdentities(t, txs[:5]))}) + sameTransactions(t, txs[:5], s.current.origin) + sameCopies(t, txs[:5], exec.transactions) + payerNow, err := exec.slotCtx.GetAccountShared(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Less(t, payerNow.Lamports, uint64(3_200_000), "the overlay saw the executed prefix") + + if reason == "sigverify_timeout" { + s.waitVerificationFn = func(context.Context, *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) { + return nil, false, context.DeadlineExceeded + } + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, Batch: turbine.NewDetachedStreamBatch(gen, 6, 8, txs[5:], verifiedIdentities(t, txs[5:]))}) + require.Nil(t, s.current) + } else { + s.discard(reason) + } + require.Empty(t, tail.added, "a discarded stream commits nothing") + durablePayer, err := env.durable.GetAccountWithoutLock(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Equal(t, uint64(3_200_000), durablePayer.Lamports, "the durable view is untouched") + + again := lifecycleWholeBlock(t, env, txs, 2) + requireSameLifecycleOutcome(t, whole, again) + + }) + } +} + +// V0 / address-lookup-table coverage. Resolving lookups mutates the message +// object (SetAddressTables refuses a second call; ResolveLookups appends to +// AccountKeys), so a stream must execute its own copies and leave the block's +// objects for the whole-block path — which may run them against a different +// parent, with a different table. + +var lifecycleTableKey = solana.PublicKey{0x7A, 0xB1, 0xE0} + +// lookupTableAccount is an active address lookup table whose only entry is +// dest, encoded the way the ALT program stores it. +func lookupTableAccount(t *testing.T, dest solana.PublicKey) *accounts.Account { + t.Helper() + var buf bytes.Buffer + enc := bin.NewBinEncoder(&buf) + require.NoError(t, enc.WriteUint32(sealevel.AddressLookupTableProgramStateLookupTable, bin.LE)) + authority := txfixture.PayerPubkey() + meta := sealevel.LookupTableMeta{DeactivationSlot: ^uint64(0), LastExtendedSlot: 1, Authority: &authority} + require.NoError(t, meta.MarshalWithEncoder(enc)) + require.NoError(t, enc.WriteBytes(dest[:], false)) + require.Equal(t, sealevel.AddressLookupTableMetaSize+32, buf.Len()) + return &accounts.Account{Key: lifecycleTableKey, Lamports: 1_000_000, Owner: addresses.AddressLookupTableAddr, Data: buf.Bytes(), RentEpoch: ^uint64(0)} +} + +func systemTransferData(lamports uint64) []byte { + data := make([]byte, 12) + binary.LittleEndian.PutUint32(data, 2) // SystemInstruction::Transfer + binary.LittleEndian.PutUint64(data[4:], lamports) + return data +} + +// signedV0TransferViaTableWire is a signed v0 transfer from the fixture payer +// to entry 0 of lifecycleTableKey (a writable lookup): static keys are the +// payer and the System program, so the destination is account index 2. +func signedV0TransferViaTableWire(t *testing.T, seq uint64) []byte { + t.Helper() + msg := solana.Message{ + Header: solana.MessageHeader{NumRequiredSignatures: 1, NumReadonlyUnsignedAccounts: 1}, + AccountKeys: solana.PublicKeySlice{txfixture.PayerPubkey(), solana.SystemProgramID}, + RecentBlockhash: txfixture.TestBlockhash(), + Instructions: []solana.CompiledInstruction{{ + ProgramIDIndex: 1, + Accounts: []uint16{0, 2}, + Data: systemTransferData(1_000 + seq), + }}, + AddressTableLookups: solana.MessageAddressTableLookupSlice{{AccountKey: lifecycleTableKey, WritableIndexes: []uint8{0}}}, + } + _, err := msg.SetVersion(solana.MessageVersionV0) + require.NoError(t, err) + tx := &solana.Transaction{Message: msg} + payerKey := txfixture.PayerPrivateKey() + _, err = tx.Sign(func(key solana.PublicKey) *solana.PrivateKey { + if key.Equals(txfixture.PayerPubkey()) { + return &payerKey + } + return nil + }) + require.NoError(t, err) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + return wire +} + +// decodeTransactions decodes wires the way turbine does, so each call yields +// fresh, unresolved objects. +func decodeTransactions(t *testing.T, wires [][]byte) []*solana.Transaction { + t.Helper() + txs := make([]*solana.Transaction, len(wires)) + for i, wire := range wires { + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + require.Equal(t, solana.MessageVersionV0, tx.Message.GetVersion()) + require.False(t, tx.Message.IsResolved()) + txs[i] = tx + } + return txs +} + +// newLifecycleEnvWithTable is newLifecycleEnv with a well-funded payer, the +// lookup table pointing at dest, and dest as an existing rent-exempt account, +// so every v0 transfer succeeds and the credited destination shows which +// table resolved it. +func newLifecycleEnvWithTable(t *testing.T, dest solana.PublicKey) *lifecycleEnv { + t.Helper() + env := newLifecycleEnv(t) + _ = env.durable.SetAccountWithoutLock(txfixture.PayerPubkey(), &accounts.Account{ + Key: txfixture.PayerPubkey(), Lamports: 10_000_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: ^uint64(0), + }) + _ = env.durable.SetAccountWithoutLock(dest, &accounts.Account{ + Key: dest, Lamports: 10_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: ^uint64(0), + }) + _ = env.durable.SetAccountWithoutLock(lifecycleTableKey, lookupTableAccount(t, dest)) + return env +} + +// A stream refuses a batch whose transactions already carry resolution: +// their account keys were derived against a parent the stream cannot vouch +// for. The assembler never produces one; this pins the refusal. +func TestStreamingRefusesResolvedInput(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + txs := decodeTransactions(t, [][]byte{signedV0TransferViaTableWire(t, 1), signedV0TransferViaTableWire(t, 2)}) + identities := verifiedIdentities(t, txs) + require.NoError(t, txs[1].Message.SetAddressTables(map[solana.PublicKey]solana.PublicKeySlice{lifecycleTableKey: {{0xD1}}})) + require.NoError(t, txs[1].Message.ResolveLookups()) + + h := newStreamingTestHarness(t) + h.exec.handleEvent(h.event(turbine.NewDetachedStreamBatch(h.gen, 1, 2, txs, identities))) + require.Nil(t, h.exec.current) + require.Equal(t, "resolved_input", h.discardReason()) + require.False(t, txs[0].Message.IsResolved(), "the unresolved sibling is untouched") +} + +func TestStreamingLifecycleV0LookupsMatchWholeBlock(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + dest := solana.PublicKey{0xDA} + wires := make([][]byte, 6) + for i := range wires { + wires[i] = signedV0TransferViaTableWire(t, uint64(i)) + } + whole := lifecycleWholeBlock(t, newLifecycleEnvWithTable(t, dest), decodeTransactions(t, wires), 2) + require.Contains(t, whole.delta, dest) + require.Equal(t, uint64(10_000_000+6*1_000+0+1+2+3+4+5), whole.delta[dest].Lamports, "every transfer reached the table's entry") + + orig := decodeTransactions(t, wires) + streamed, _ := lifecycleStream(t, newLifecycleEnvWithTable(t, dest), orig, []int{2}, 2, 2) + requireSameLifecycleOutcome(t, whole, streamed) + for i := 0; i < 4; i++ { + require.False(t, orig[i].Message.IsResolved(), "streamed transaction %d ran as a copy", i) + } + for i := 4; i < 6; i++ { + require.True(t, orig[i].Message.IsResolved(), "suffix transaction %d ran as the block's own object, like whole-block execution", i) + } +} + +// A v0 prefix executed on a stream against parent A is discarded; the same +// authoritative objects then execute whole against parent B, whose table +// names a different destination. The block's objects must still resolve +// (they were never touched), B's destination must be the one credited, and +// the result must equal a fresh-decoded whole-block reference on B. +func TestStreamingLifecycleDiscardedV0PrefixResolvesAgainstTheNewParent(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + destA, destB := solana.PublicKey{0xA1}, solana.PublicKey{0xB2} + wires := make([][]byte, 6) + for i := range wires { + wires[i] = signedV0TransferViaTableWire(t, uint64(10+i)) + } + orig := decodeTransactions(t, wires) + + envA := newLifecycleEnvWithTable(t, destA) + tail := &lifecycleTail{durable: envA.durable} + statuses := NewTransactionStatusCache() + exec := newBlockExecution(envA.acctsDb, envA.block(nil), envA.epochSchedule, 2, nil, &persistedTracker{}, tail, statuses, false, envA.parent) + require.NoError(t, exec.open()) + feed := newFakeStreamFeed() + gen := turbine.NewDetachedStreamGeneration(lifecycleSlot) + feed.status[gen] = turbine.StreamActive + s := newStreamingExecutor(streamingDeps{feed: feed, epochSchedule: envA.epochSchedule, tail: tail, transactionStatuses: statuses}) + s.current = &streamingSlot{slot: lifecycleSlot, generation: gen, parentSlot: lifecycleParentSlot, parentID: lifecycleParentBlockID, + exec: exec, pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: time.Now(), nextStart: 1} + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, + Batch: turbine.NewDetachedStreamBatch(gen, 1, 4, orig[:4], verifiedIdentities(t, orig[:4]))}) + require.NotNil(t, s.current, "stream discarded: %s", metrics.GlobalBlockReplay.StreamingExecution.DiscardReason) + sameTransactions(t, orig[:4], s.current.origin) + sameCopies(t, orig[:4], exec.transactions) + creditedA, err := exec.slotCtx.GetAccountShared(destA) + require.NoError(t, err) + require.Equal(t, uint64(10_000_000+4*1_000+10+11+12+13), creditedA.Lamports, "the stream resolved through A's table") + for i, tx := range orig { + require.False(t, tx.Message.IsResolved(), "block object %d must be untouched by the stream", i) + } + + s.discard("fork_switch") + require.Empty(t, tail.added) + + envB := newLifecycleEnvWithTable(t, destB) + whole := lifecycleWholeBlock(t, envB, orig, 2) + require.Contains(t, whole.delta, destB, "the block's objects resolved through B's table") + require.NotContains(t, whole.delta, destA, "nothing of A's resolution survived") + require.Equal(t, uint64(10_000_000+6*1_000+10+11+12+13+14+15), whole.delta[destB].Lamports) + for i, tx := range orig { + require.True(t, tx.Message.IsResolved(), "block object %d was resolved by the whole-block path", i) + } + + reference := lifecycleWholeBlock(t, newLifecycleEnvWithTable(t, destB), decodeTransactions(t, wires), 2) + requireSameLifecycleOutcome(t, reference, whole) +} diff --git a/pkg/replay/streaming_program_mix_test.go b/pkg/replay/streaming_program_mix_test.go new file mode 100644 index 000000000..3391eb15f --- /dev/null +++ b/pkg/replay/streaming_program_mix_test.go @@ -0,0 +1,193 @@ +package replay + +import ( + "bytes" + "encoding/binary" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + bin "github.com/gagliardetto/binary" + "github.com/gagliardetto/solana-go" + "github.com/gagliardetto/solana-go/programs/system" + "github.com/stretchr/testify/require" +) + +func signLifecycleInstructions(t *testing.T, instructions ...solana.Instruction) *solana.Transaction { + t.Helper() + tx, err := solana.NewTransaction(instructions, txfixture.TestBlockhash(), solana.TransactionPayer(txfixture.PayerPubkey())) + require.NoError(t, err) + key := txfixture.PayerPrivateKey() + _, err = tx.Sign(func(pk solana.PublicKey) *solana.PrivateKey { + if pk == key.PublicKey() { + return &key + } + return nil + }) + require.NoError(t, err) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + decoded, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + return decoded +} + +// Each case has a real write consumed or replaced in a subsequent group. Fresh +// wire decoding avoids accidentally sharing resolved transactions across paths. +func TestStreamingLifecycleProgramMutations(t *testing.T) { + previous := StreamingExecutionCfg + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + t.Cleanup(func() { StreamingExecutionCfg = previous }) + payer := txfixture.PayerPubkey() + key := solana.PublicKey{0xC1} + previousVote := global.VoteCacheItem(key) + t.Cleanup(func() { + if previousVote == nil { + global.DeleteVoteCacheItem(key) + } else { + global.PutVoteCacheItem(key, previousVote) + } + }) + for _, kind := range []string{"nonce", "lookup extension", "vote commission", "program upgrade", "program deployment"} { + t.Run(kind, func(t *testing.T) { + setup := func() (*lifecycleEnv, []*solana.Transaction) { + env := newLifecycleEnv(t) + env.acctsDb.InitCaches() + t.Cleanup(env.acctsDb.ProgramCache.Close) + t.Cleanup(env.acctsDb.VoteAcctCache.Close) + t.Cleanup(env.acctsDb.CommonAcctsCache.Close) + put := func(pk solana.PublicKey, owner solana.PublicKey, data []byte, executable bool) { + require.NoError(t, env.durable.SetAccountWithoutLock(pk, &accounts.Account{Key: pk, Owner: owner, Lamports: 100_000_000, Data: data, Executable: executable, RentEpoch: ^uint64(0)})) + } + put(payer, addresses.SystemProgramAddr, nil, false) + native := func(pk solana.PublicKey) { put(pk, addresses.NativeLoaderAddr, nil, true) } + var txs []*solana.Transaction + switch kind { + case "nonce": + state := sealevel.NonceStateVersions{Type: sealevel.NonceVersionCurrent, Current: sealevel.NonceData{IsInitialized: true, Authority: payer, DurableNonce: [32]byte{0xAA}, FeeCalculator: sealevel.FeeCalculator{LamportsPerSignature: 5000}}} + data, err := state.Marshal() + require.NoError(t, err) + put(key, addresses.SystemProgramAddr, data, false) + // Repeated advances have different messages but the same nonce account; + // only the first may advance in this bank. Later instructions must see it. + for i := uint64(1); i <= 3; i++ { + txs = append(txs, signLifecycleInstructions(t, system.NewAdvanceNonceAccountInstruction(key, solana.SysVarRecentBlockHashesPubkey, payer).Build(), system.NewTransferInstruction(i, payer, txfixture.DestPubkey()).Build())) + } + case "lookup extension": + native(addresses.AddressLookupTableAddr) + table := lookupTableAccount(t, txfixture.DestPubkey()) + table.Lamports = 100_000_000 + require.NoError(t, env.durable.SetAccountWithoutLock(lifecycleTableKey, table)) + for i := byte(1); i <= 3; i++ { + instruction := sealevel.AddrLookupTableInstrExtendLookupTable{NewAddresses: []solana.PublicKey{{0xD2, i}}} + var b bytes.Buffer + require.NoError(t, instruction.MarshalWithEncoder(bin.NewBinEncoder(&b))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.AddressLookupTableAddr, solana.AccountMetaSlice{solana.Meta(lifecycleTableKey).WRITE(), solana.Meta(payer).SIGNER()}, b.Bytes()))) + } + case "vote commission": + global.DeleteVoteCacheItem(key) + native(addresses.VoteProgramAddr) + state := sealevel.VoteStateVersions{Type: sealevel.VoteStateVersionCurrent, Current: sealevel.VoteState{AuthorizedWithdrawer: payer, Commission: 30}} + var b bytes.Buffer + require.NoError(t, state.MarshalWithEncoder(bin.NewBinEncoder(&b))) + data := make([]byte, sealevel.VoteStateV3Size) + copy(data, b.Bytes()) + put(key, addresses.VoteProgramAddr, data, false) + for _, commission := range []byte{20, 10, 5} { + data := binary.LittleEndian.AppendUint32(nil, sealevel.VoteProgramInstrTypeUpdateCommission) + data = append(data, commission) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.VoteProgramAddr, solana.AccountMetaSlice{solana.Meta(key).WRITE(), solana.Meta(payer).SIGNER()}, data))) + } + case "program upgrade", "program deployment": + native(addresses.BpfLoaderUpgradeableAddr) + programDataKey := solana.PublicKey{0xC2} + bufferKey := solana.PublicKey{0xC3} + elf := fixtures.Load(t, "sbpf", "noop_aligned.so") + encode := func(state sealevel.UpgradeableLoaderState, size int) []byte { + var b bytes.Buffer + require.NoError(t, state.MarshalWithEncoder(bin.NewBinEncoder(&b))) + out := make([]byte, size) + copy(out, b.Bytes()) + return out + } + put(key, addresses.BpfLoaderUpgradeableAddr, encode(sealevel.UpgradeableLoaderState{Type: sealevel.UpgradeableLoaderStateTypeProgram, Program: sealevel.UpgradeableLoaderStateProgram{ProgramDataAddress: programDataKey}}, 36), true) + pd := encode(sealevel.UpgradeableLoaderState{Type: sealevel.UpgradeableLoaderStateTypeProgramData, ProgramData: sealevel.UpgradeableLoaderStateProgramData{Slot: 1, UpgradeAuthorityAddress: &payer}}, 45+len(elf)) + copy(pd[45:], elf) + put(programDataKey, addresses.BpfLoaderUpgradeableAddr, pd, false) + buf := encode(sealevel.UpgradeableLoaderState{Type: sealevel.UpgradeableLoaderStateTypeBuffer, Buffer: sealevel.UpgradeableLoaderStateBuffer{AuthorityAddress: &payer}}, 37+len(elf)) + copy(buf[37:], elf) + put(bufferKey, addresses.BpfLoaderUpgradeableAddr, buf, false) + if kind == "program deployment" { + programDataKey, _, err := solana.FindProgramAddress([][]byte{key[:]}, addresses.BpfLoaderUpgradeableAddr) + require.NoError(t, err) + put(key, addresses.BpfLoaderUpgradeableAddr, make([]byte, 36), false) + // No pre-funded PDA: Deploy creates it with a signed CPI. + require.NoError(t, env.durable.SetAccountWithoutLock(programDataKey, &accounts.Account{Key: programDataKey, Owner: addresses.SystemProgramAddr})) + write := sealevel.UpgradeableLoaderInstrWrite{Offset: 0, Bytes: elf[:8]} + var b bytes.Buffer + require.NoError(t, write.MarshalWithEncoder(bin.NewBinEncoder(&b))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.BpfLoaderUpgradeableAddr, solana.AccountMetaSlice{solana.Meta(bufferKey).WRITE(), solana.Meta(payer).SIGNER()}, b.Bytes()))) + deploy := sealevel.UpgradeableLoaderInstrDeployWithMaxDataLen{MaxDataLen: uint64(len(elf))} + b.Reset() + require.NoError(t, deploy.MarshalWithEncoder(bin.NewBinEncoder(&b))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.BpfLoaderUpgradeableAddr, solana.AccountMetaSlice{solana.Meta(payer).WRITE().SIGNER(), solana.Meta(programDataKey).WRITE(), solana.Meta(key).WRITE(), solana.Meta(bufferKey).WRITE(), solana.Meta(sealevel.SysvarRentAddr), solana.Meta(sealevel.SysvarClockAddr), solana.Meta(addresses.SystemProgramAddr), solana.Meta(payer).SIGNER()}, b.Bytes()))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(key, nil, []byte{2}))) + break + } + + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(key, nil, []byte{1}))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.BpfLoaderUpgradeableAddr, solana.AccountMetaSlice{solana.Meta(programDataKey).WRITE(), solana.Meta(key).WRITE(), solana.Meta(bufferKey).WRITE(), solana.Meta(payer).WRITE(), solana.Meta(sealevel.SysvarRentAddr), solana.Meta(sealevel.SysvarClockAddr), solana.Meta(payer).SIGNER()}, binary.LittleEndian.AppendUint32(nil, sealevel.UpgradeableLoaderInstrTypeUpgrade)))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(key, nil, []byte{2}))) + } + return env, txs + } + env, txs := setup() + whole := lifecycleWholeBlock(t, env, txs, 2) + switch kind { + case "nonce": + require.Contains(t, whole.delta, key) + state, err := sealevel.UnmarshalNonceStateVersions(whole.delta[key].Data) + require.NoError(t, err) + require.NotEqual(t, [32]byte{0xAA}, state.State().DurableNonce) + case "lookup extension": + require.Contains(t, whole.delta, lifecycleTableKey) + require.Len(t, whole.delta[lifecycleTableKey].Data, sealevel.AddressLookupTableMetaSize+4*32) + case "vote commission": + require.Contains(t, whole.delta, key) + state, err := sealevel.UnmarshalVersionedVoteState(whole.delta[key].Data) + require.NoError(t, err) + require.Equal(t, byte(5), state.ConvertToCurrent().Commission) + case "program upgrade", "program deployment": + pk := solana.PublicKey{0xC2} + if kind == "program deployment" { + var err error + pk, _, err = solana.FindProgramAddress([][]byte{key[:]}, addresses.BpfLoaderUpgradeableAddr) + require.NoError(t, err) + require.True(t, whole.delta[key].Executable) + } + require.Contains(t, whole.delta, pk) + state, err := sealevel.UnmarshalUpgradeableLoaderState(whole.delta[pk].Data) + require.NoError(t, err) + require.Equal(t, lifecycleSlot, state.ProgramData.Slot) + } + for _, workers := range []int{1, 4} { + for suffix := 0; suffix <= 3; suffix++ { + t.Run(fmt.Sprintf("workers%d/suffix%d", workers, suffix), func(t *testing.T) { + env, txs := setup() + var splits []int + for i := 1; i < 3-suffix; i++ { + splits = append(splits, i) + } + got, _ := lifecycleStream(t, env, txs, splits, suffix, workers) + requireSameLifecycleOutcome(t, whole, got) + }) + } + } + }) + } +} diff --git a/pkg/replay/streaming_realfeed_test.go b/pkg/replay/streaming_realfeed_test.go new file mode 100644 index 000000000..2c86f7b60 --- /dev/null +++ b/pkg/replay/streaming_realfeed_test.go @@ -0,0 +1,494 @@ +package replay + +import ( + "context" + "net" + "testing" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/sigverify" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// The real-feed gate. A leader-side BroadcastSession shreds a header, entry +// batches (legacy and v0 transfers), a footer and the ending tick; the +// packets cross a loopback UDP socket into a real UDPReceiver (assembler, +// entry prefetch, signature verifier); the receiver's streaming feed drives +// the real streamingExecutor, which opens a bank on the lifecycle +// environment and executes the prefix while the slot is still incomplete; +// the receiver then emits the complete block, finalize matches it, and the +// result must equal whole-block replay of the same block — enforced twice: +// by the footer's expected bank hash inside finalize (the footer carries the +// whole-block reference hash) and by the explicit outcome comparison. + +const ( + realFeedSlot = lifecycleSlot + realFeedParentSlot = lifecycleParentSlot + realFeedShredVersion = uint16(7) + realFeedTickHashByte = 0xEE + realFeedDriveDeadline = 20 * time.Second +) + +// receiverFeed is the block source's view of the feed, backed by a receiver. +type receiverFeed struct { + r *turbine.UDPReceiver + events chan turbine.StreamEvent +} + +func (f *receiverFeed) StreamEvents() <-chan turbine.StreamEvent { return f.events } +func (f *receiverFeed) StreamStatusOf(g turbine.StreamGeneration) turbine.StreamStatus { + return f.r.StreamStatusOf(g) +} +func (f *receiverFeed) PendingStreamBatches(g turbine.StreamGeneration, from uint32) []*turbine.StreamBatch { + return f.r.PendingStreamBatches(g, from) +} +func (f *receiverFeed) PrioritizeStreamRepair(turbine.StreamGeneration) {} + +type realFeedRig struct { + slot uint64 + t *testing.T + env *lifecycleEnv + receiver *turbine.UDPReceiver + feed *receiverFeed + broadcaster *turbine.UDPBroadcaster + leader solana.PrivateKey + lastCtx *sealevel.SlotCtx + tail *lifecycleTail + statuses *TransactionStatusCache + exec *streamingExecutor + block *b.Block + // mark is the loop's record of the executed parent (replayed before any + // shred of the child is broadcast), as the loop would hold it. + mark streamingFrontierMark + // the slot's content, as wire bytes, decoded fresh for every use + legacyWires, v0Wires [][]byte +} + +func newRealFeedRig(t *testing.T, dest solana.PublicKey, eventBuffer int) *realFeedRig { + t.Helper() + previousOverlap := sigverify.Cfg.DisableShredOverlap + sigverify.Cfg.DisableShredOverlap = false // the entry prefetch is what streams + t.Cleanup(func() { sigverify.Cfg.DisableShredOverlap = previousOverlap }) + // The open-age bound is the loop's protection against a stalled slot; a + // loaded CI host must not trip it between two drive calls. + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, Workers: 2, MaxOpenAge: realFeedDriveDeadline} + t.Cleanup(func() { StreamingExecutionCfg = StreamingExecutionConfig{} }) + // The per-block collector, as the loop resets it before every wait. + previousMetrics := metrics.GlobalBlockReplay + metrics.GlobalBlockReplay = metrics.BlockReplay{} + t.Cleanup(func() { metrics.GlobalBlockReplay = previousMetrics }) + + env := newLifecycleEnvWithTable(t, dest) + rig := &realFeedRig{slot: realFeedSlot, t: t, env: env, leader: solana.NewWallet().PrivateKey} + for i := 0; i < 4; i++ { + rig.legacyWires = append(rig.legacyWires, txfixture.MustSignedTransferWire(uint64(2000+i))) + } + for i := 0; i < 3; i++ { + rig.v0Wires = append(rig.v0Wires, signedV0TransferViaTableWire(t, uint64(30+i))) + } + + // The executed parent, as the loop would hold it: its bank hash is the + // child's ParentBankhash and its sysvar snapshot the child's parent. + rig.lastCtx = &sealevel.SlotCtx{ + Slot: realFeedParentSlot, + Epoch: 0, + FinalBankhash: append([]byte{0x88}, make([]byte, 31)...), + FeeRateGovernor: &sealevel.FeeRateGovernor{TargetLamportsPerSignature: 5_000, LamportsPerSignature: 5_000}, + VoteTimestamps: map[solana.PublicKey]sealevel.BlockTimestamp{}, + } + require.NoError(t, rig.lastCtx.PublishBankSysvars(env.parent)) + + // A real receiver on a loopback port we pick ourselves (the receiver does + // not report an ephemeral bind), fed by a real broadcaster. + probe, err := net.ListenPacket("udp", "127.0.0.1:0") + require.NoError(t, err) + bindAddr := probe.LocalAddr().String() + require.NoError(t, probe.Close()) + receiver := turbine.NewUDPReceiver(bindAddr) + receiver.SetShredVersion(realFeedShredVersion) + leaderKey := rig.leader.PublicKey() + receiver.SetLeaderForSlot(func(uint64) (solana.PublicKey, bool) { return leaderKey, true }) + rig.feed = &receiverFeed{r: receiver, events: make(chan turbine.StreamEvent, eventBuffer)} + receiver.SubscribeStream(rig.feed.events) + ctx, cancel := context.WithCancel(context.Background()) + runDone := make(chan struct{}) + go func() { _ = receiver.Run(ctx); close(runDone) }() + t.Cleanup(func() { + cancel() + select { + case <-runDone: + case <-time.After(5 * time.Second): + t.Error("receiver did not stop") + } + }) + select { + case err := <-receiver.Ready(): + require.NoError(t, err) + case <-time.After(5 * time.Second): + t.Fatal("receiver did not become ready") + } + rig.receiver = receiver + udpAddr, err := net.ResolveUDPAddr("udp", bindAddr) + require.NoError(t, err) + rig.broadcaster, err = turbine.NewUDPBroadcaster("127.0.0.1:0") + require.NoError(t, err) + rig.broadcaster.AddPeer(udpAddr) + t.Cleanup(func() { _ = rig.broadcaster.Close() }) + + rig.tail = &lifecycleTail{durable: env.durable} + rig.statuses = NewTransactionStatusCache() + now := time.Now() + rig.mark = streamingFrontierMark{slot: realFeedParentSlot, fullNanos: now.Add(-10 * time.Millisecond).UnixNano(), admittedAt: now.Add(-5 * time.Millisecond), replayedAt: now, waitEnteredAt: now.Add(time.Microsecond)} + rig.exec = newStreamingExecutor(streamingDeps{ + acctsDb: env.acctsDb, + feed: rig.feed, + epochSchedule: env.epochSchedule, + txParallelism: 2, + persistedHashes: &persistedTracker{}, + tail: rig.tail, + transactionStatuses: rig.statuses, + alpenglowMode: true, + unrootedTailUsed: true, + lastSlotCtx: func() *sealevel.SlotCtx { return rig.lastCtx }, + frontier: func() uint64 { return realFeedParentSlot }, + frontierMark: func() streamingFrontierMark { return rig.mark }, + currentFeatures: func() *features.Features { return env.feats }, + currentEpoch: func() uint64 { return 0 }, + rewardsInFlight: func() bool { return false }, + switchPending: func() bool { return false }, + executedBlockID: func(uint64) (solana.Hash, bool) { return lifecycleParentBlockID, true }, + }) + t.Cleanup(rig.exec.shutdown) + return rig +} + +// entries decodes the slot's content into the two entry batches the leader +// broadcasts: legacy transfers, then v0 transfers through the lookup table. +func (rig *realFeedRig) entries() (legacy, v0 []turbine.Entry) { + values := func(wires [][]byte) []solana.Transaction { + out := make([]solana.Transaction, len(wires)) + for i, tx := range decodeWires(rig.t, wires) { + out[i] = *tx + } + return out + } + return []turbine.Entry{{NumHashes: 1, Hash: solana.Hash{0x11}, Txns: values(rig.legacyWires)}}, + []turbine.Entry{{NumHashes: 1, Hash: solana.Hash{0x22}, Txns: values(rig.v0Wires)}} +} + +func (rig *realFeedRig) allWires() [][]byte { + return append(append([][]byte(nil), rig.legacyWires...), rig.v0Wires...) +} + +func decodeWires(t *testing.T, wires [][]byte) []*solana.Transaction { + t.Helper() + txs := make([]*solana.Transaction, len(wires)) + for i, wire := range wires { + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + txs[i] = tx + } + return txs +} + +// configure applies what the replay loop applies to every block on the +// executed parent before execution. +func (rig *realFeedRig) configure(block *b.Block) { + require.NoError(rig.t, configureBlockFromParent(block, rig.lastCtx, rig.env.epochSchedule, false)) + block.Epoch = rig.env.epochSchedule.GetEpoch(block.Slot) + block.Features = rig.env.feats +} + +// reference executes the slot whole, from freshly decoded objects, exactly as +// the loop would: same parent, same content, same last-entry hash. +func (rig *realFeedRig) reference() lifecycleOutcome { + block := &b.Block{ + Slot: rig.slot, + SourceParentSlot: realFeedParentSlot, + FromLiveStream: true, + AlpenglowParentBlockID: lifecycleParentBlockID, + HasAlpenglowParentBlockID: true, + Transactions: decodeWires(rig.t, rig.allWires()), + Blockhash: solana.Hash{realFeedTickHashByte}, + } + rig.configure(block) + block.MarkTransactionSignaturesVerified() + tail := &lifecycleTail{durable: rig.env.durable} + slotCtx, err := ProcessBlock(rig.env.acctsDb, block, rig.env.epochSchedule, 2, nil, &persistedTracker{}, tail, NewTransactionStatusCache(), false, rig.env.parent) + require.NoError(rig.t, err) + return lifecycleOutcomeOf(rig.t, slotCtx, tail) +} + +func (rig *realFeedRig) session() *turbine.BroadcastSession { + return turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: rig.leader, + Slot: rig.slot, + ParentSlot: realFeedParentSlot, + ParentBlockID: lifecycleParentBlockID, + ParentChainedMerkleRoot: solana.Hash{0xBB}, + Broadcaster: rig.broadcaster, + Version: realFeedShredVersion, + }) +} + +// broadcastPrefix sends the header and both entry batches; the slot stays +// incomplete (no footer, no ending tick). +func (rig *realFeedRig) broadcastPrefix(session *turbine.BroadcastSession) { + legacy, v0 := rig.entries() + require.NoError(rig.t, session.BroadcastHeader(lifecycleParentBlockID)) + require.NoError(rig.t, session.BroadcastEntryBatch(legacy)) + require.NoError(rig.t, session.BroadcastEntryBatch(v0)) +} + +// broadcastCompletion sends the footer carrying the expected bank hash and +// the ending tick, which completes the slot. +func (rig *realFeedRig) broadcastCompletion(session *turbine.BroadcastSession, expectedBankhash []byte) { + require.NoError(rig.t, session.BroadcastFooter(solana.HashFromBytes(expectedBankhash), 1_700_000_000_000_000_000, nil, nil)) + require.NoError(rig.t, session.BroadcastEndingTickLast(solana.Hash{realFeedTickHashByte})) +} + +// drive runs the replay loop's wait as the loop would: feed wake-ups and +// poll ticks go to the executor, a complete block ends the wait. +func (rig *realFeedRig) drive(until func() bool, what string) { + rig.t.Helper() + deadline := time.After(realFeedDriveDeadline) + for !until() { + select { + case event := <-rig.feed.events: + rig.exec.handleEvent(event) + case <-rig.exec.tick(): + rig.exec.handleTick() + case blk, ok := <-rig.receiver.Blocks(): + require.True(rig.t, ok, "receiver closed its block channel") + rig.receiver.AcknowledgeBlockDelivery(blk.Slot) + rig.block = blk + case <-deadline: + reason := metrics.GlobalBlockReplay.StreamingExecution.DiscardReason + rig.t.Fatalf("timed out waiting for %s (stream open: %v, discard reason %q, block: %v)", what, rig.exec.current != nil, reason, rig.block != nil) + } + } +} + +func (rig *realFeedRig) executedPrefix() int { + if rig.exec.current == nil { + return -1 + } + return len(rig.exec.current.origin) +} + +// finalizeAndCompare completes the streamed bank against the emitted block +// and checks it against the whole-block reference. +func (rig *realFeedRig) finalizeAndCompare(reference lifecycleOutcome) { + rig.t.Helper() + block := rig.block + require.NotNil(rig.t, block) + require.Equal(rig.t, rig.slot, block.Slot) + require.True(rig.t, block.HasAlpenglowParentBlockID) + require.Equal(rig.t, lifecycleParentBlockID, solana.Hash(block.AlpenglowParentBlockID)) + require.True(rig.t, block.HasExpectedBankhash, "the footer carried the reference bank hash") + require.Len(rig.t, block.Transactions, len(rig.allWires())) + require.Positive(rig.t, block.ShredFullNanos) + rig.configure(block) + + slotCtx, ok, err := rig.exec.finalize(block, rig.env.parent) + require.NoError(rig.t, err) + require.True(rig.t, ok, "the stream must accept its own block (discard reason %q)", metrics.GlobalBlockReplay.StreamingExecution.DiscardReason) + require.Nil(rig.t, rig.exec.current) + streamed := lifecycleOutcomeOf(rig.t, slotCtx, rig.tail) + requireSameLifecycleOutcome(rig.t, reference, streamed) + require.Equal(rig.t, reference.bankhash, block.ExpectedBankhash[:], "finalize verified the footer hash against the same value") + + record := metrics.GlobalBlockReplay.StreamingExecution + require.Equal(rig.t, uint64(1), record.Opened) + require.Equal(rig.t, uint64(len(rig.allWires())), record.Transactions) + require.Zero(rig.t, record.Discarded, "discard reason %q", record.DiscardReason) + require.Empty(rig.t, record.NotOpenedReason) + + // The open timeline, in order: the parent was replayed before the child's + // header was decoded, the header was seen no earlier than decoded, the + // stream opened after that, executed, and finalized after the last shred. + require.Equal(rig.t, rig.mark.replayedAt.UnixNano(), record.ParentReplayedNanos) + require.Equal(rig.t, rig.mark.admittedAt.UnixNano(), record.ParentAdmittedNanos) + require.Equal(rig.t, rig.mark.fullNanos, record.ParentFullNanos) + require.Less(rig.t, record.ParentReplayedNanos, record.HeaderReadyNanos) + require.LessOrEqual(rig.t, record.HeaderReadyNanos, record.HeaderSeenNanos) + require.LessOrEqual(rig.t, record.HeaderSeenNanos, record.OpenedNanos) + require.LessOrEqual(rig.t, record.OpenedNanos, record.FirstGroupStartNanos) + require.Equal(rig.t, block.ShredFullNanos, record.FullNanos) + require.LessOrEqual(rig.t, record.FullNanos, record.FinalizeStartNanos) + require.Zero(rig.t, record.OpenWaitParentArrival.Count, "the parent was fully received before the header") + require.Zero(rig.t, record.OpenWaitParentReplay.Count, "the parent was replayed before the header") + require.Zero(rig.t, record.OpenWaitParentPreAdmission.Count) + require.Zero(rig.t, record.OpenWaitParentPostAdmission.Count) + require.Equal(rig.t, uint64(1), record.OpenWaitLoop.Count) + require.Equal(rig.t, uint64(record.OpenedNanos-record.HeaderReadyNanos), record.OpenWaitLoop.SumNanoseconds, "with nothing to wait for, the whole open delay is the loop's") + require.Equal(rig.t, rig.mark.waitEnteredAt.UnixNano(), record.WaitEnteredNanos) + // Every group ran before the completion was even broadcast. + require.Positive(rig.t, record.LastGroupEndNanos) + require.LessOrEqual(rig.t, record.LastGroupEndNanos, record.FullNanos) + require.Zero(rig.t, record.TxLoopAfterFull.Count) + require.Zero(rig.t, record.GroupsStraddlingFull) + require.Contains(rig.t, []uint64{uint64(len(rig.legacyWires)), uint64(len(rig.allWires()))}, record.LargestGroupTransactions, "one or two groups, depending on how the batches were pulled") + require.Zero(rig.t, record.OpenWaitPostReplay.Count, "the header arrived after the loop was already waiting") + require.Equal(rig.t, record.OpenWaitLoop, record.OpenWaitDispatch, "…so the loop's delay is all dispatch") +} + +func TestStreamingRealFeedExecutesPrefixBeforeCompletionAndMatchesWholeBlock(t *testing.T) { + dest := solana.PublicKey{0xF1} + rig := newRealFeedRig(t, dest, 64) + reference := rig.reference() + require.Contains(t, reference.delta, dest) + total := len(rig.allWires()) + + session := rig.session() + rig.broadcastPrefix(session) + rig.drive(func() bool { return rig.executedPrefix() == total }, "the prefix to execute") + require.False(t, rig.receiver.SlotCompleted(realFeedSlot), "the slot is still incomplete while the prefix executes") + require.Nil(t, rig.block) + prefixDone := time.Now() + for i, tx := range rig.exec.current.origin[len(rig.legacyWires):] { + require.Equal(t, solana.MessageVersionV0, tx.Message.GetVersion()) + require.False(t, tx.Message.IsResolved(), "block object %d ran as a stream-owned copy", i) + } + sameCopies(t, rig.exec.current.origin, rig.exec.current.exec.transactions) + + rig.broadcastCompletion(session, reference.bankhash) + rig.drive(func() bool { return rig.block != nil }, "the complete block") + require.NotNil(t, rig.exec.current, "the stream survives completion") + require.Equal(t, turbine.StreamDone, rig.receiver.StreamStatusOf(rig.exec.current.generation)) + require.True(t, time.Unix(0, rig.block.ShredFullNanos).After(prefixDone), "the prefix executed before the last shred arrived") + for i, tx := range rig.exec.current.origin { + require.Same(t, rig.block.Transactions[i], tx, "the executed prefix is the block, by identity") + } + + retainedLoader := rig.exec.current.exec.accountLoader + require.Positive(t, retainedLoader.SourceBatch.Count) + // Replay resets the global collector while waiting for the full block. + // Early loader work must survive and be published exactly once. + metrics.GlobalBlockReplay.AccountLoader = metrics.AccountLoader{} + rig.finalizeAndCompare(reference) + require.Equal(t, retainedLoader.SourceBatch, metrics.GlobalBlockReplay.AccountLoader.SourceBatch) + require.Equal(t, retainedLoader.RequestedKeys, metrics.GlobalBlockReplay.AccountLoader.RequestedKeys) + require.Equal(t, retainedLoader.ParentAccounts, metrics.GlobalBlockReplay.AccountLoader.ParentAccounts) + require.Positive(t, metrics.GlobalBlockReplay.StreamingExecution.TxLoopBeforeFull.Count, "transaction work finished before the slot was full") + require.Zero(t, rig.receiver.StreamDroppedEvents()) +} + +// Dropped wake-ups: with room for one event, whichever of the three prefix +// wake-ups (header, two batches) the prefetch publishes first is queued and +// the other two are dropped — the prefetch decodes ranges concurrently, so +// the survivor is not fixed. The executor must reach the same state from any +// of them: a surviving header opens the stream and pulls the batches; a +// surviving batch recovers the header from the assembler, then opens and +// pulls the same way. Nothing is read from the feed until every wake-up has +// been published, so exactly two are dropped. +func TestStreamingRealFeedRecoversDroppedWakeups(t *testing.T) { + dest := solana.PublicKey{0xF2} + rig := newRealFeedRig(t, dest, 1) + reference := rig.reference() + total := len(rig.allWires()) + + session := rig.session() + rig.broadcastPrefix(session) + require.Eventually(t, func() bool { + return rig.receiver.StreamDroppedEvents() == 2 + }, realFeedDriveDeadline, 5*time.Millisecond, "one wake-up queued, two dropped") + var survivor turbine.StreamEvent + select { + case survivor = <-rig.feed.events: + default: + t.Fatal("the surviving wake-up is not queued") + } + var live bool + survivor, live = survivor.Resolve() + require.True(t, live) + require.Zero(t, len(rig.feed.events), "nothing else was published") + require.Equal(t, turbine.StreamBatchReady, survivor.Kind) + require.Equal(t, uint64(realFeedSlot), survivor.Slot) + require.Len(t, rig.receiver.PendingStreamBatches(survivor.Generation, 0), 3, "header and both batches are decoded and discoverable") + require.Nil(t, rig.exec.current) + + rig.exec.handleEvent(survivor) + require.NotNil(t, rig.exec.current, "the surviving wake-up (marker %v at shred %d) opened the stream", survivor.Batch.Marker, survivor.Batch.Start) + require.Equal(t, total, rig.executedPrefix(), "opening pulled every decoded batch from the assembler") + require.Equal(t, uint64(2), rig.receiver.StreamDroppedEvents(), "recovery reads the assembler, it does not replay wake-ups") + require.False(t, rig.receiver.SlotCompleted(realFeedSlot)) + + rig.broadcastCompletion(session, reference.bankhash) + rig.drive(func() bool { return rig.block != nil }, "the complete block") + rig.finalizeAndCompare(reference) +} + +// Reset while streaming: the generation is cancelled, the stream discards +// and leaves nothing behind; the slot re-broadcast is a new generation, which +// opens a new stream and finalizes to the same result. +func TestStreamingRealFeedResetDiscardsAndRenews(t *testing.T) { + dest := solana.PublicKey{0xF3} + rig := newRealFeedRig(t, dest, 64) + reference := rig.reference() + total := len(rig.allWires()) + + stakeBefore := len(global.PendingStakeEntriesSnapshot()) + rig.broadcastPrefix(rig.session()) + rig.drive(func() bool { return rig.executedPrefix() == total }, "the first prefix to execute") + firstGeneration := rig.exec.current.generation + + rig.receiver.ResetSlot(realFeedSlot) + rig.drive(func() bool { return rig.exec.current == nil }, "the cancellation") + // The reset reaches the executor either as the feed's cancellation wake-up + // or, when the poll tick is selected first, as the generation reading gone. + require.Contains(t, []string{"cancelled:reset", "gone"}, metrics.GlobalBlockReplay.StreamingExecution.DiscardReason) + require.Equal(t, turbine.StreamGone, rig.receiver.StreamStatusOf(firstGeneration)) + require.Empty(t, rig.tail.added, "a discarded stream commits nothing") + require.Len(t, global.PendingStakeEntriesSnapshot(), stakeBefore) + durablePayer, err := rig.env.durable.GetAccountWithoutLock(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Equal(t, uint64(10_000_000_000), durablePayer.Lamports, "the durable view is untouched") + // The loop starts the next replay attempt with a fresh collector. + metrics.GlobalBlockReplay = metrics.BlockReplay{} + + session := rig.session() + rig.broadcastPrefix(session) + rig.drive(func() bool { return rig.executedPrefix() == total }, "the renewed prefix to execute") + // Compared as values (a deep comparison would walk the assembler's live + // slot state without its lock). + require.False(t, firstGeneration == rig.exec.current.generation, "a re-assembled slot is a new generation") + rig.broadcastCompletion(session, reference.bankhash) + rig.drive(func() bool { return rig.block != nil }, "the complete block") + rig.finalizeAndCompare(reference) +} + +// The child executes on parent 7 while slots 8..11 are unresolved. Consuming +// those skips advances only the frontier; the completed child must still +// produce exactly the whole-block bank hash, accounts, fees and CU. +func TestStreamingRealFeedAcrossSkippedSlots(t *testing.T) { + rig := newRealFeedRig(t, solana.PublicKey{0xF1}, 64) + rig.slot += 4 + frontier := uint64(realFeedParentSlot) + rig.exec.deps.frontier = func() uint64 { return frontier } + reference := rig.reference() + session := rig.session() + rig.broadcastPrefix(session) + rig.drive(func() bool { return rig.executedPrefix() == len(rig.allWires()) }, "prefix across unresolved skips") + require.Equal(t, uint64(realFeedParentSlot), frontier, "speculation does not certify skips or advance replay") + require.False(t, rig.receiver.SlotCompleted(rig.slot)) + cur := rig.exec.current + for slot := frontier + 1; slot < rig.slot; slot++ { + rig.exec.beforeBlock(&b.Block{Slot: slot, IsSkipped: true}) + rig.exec.discardSlot(slot, "skipped") + frontier = slot + rig.exec.handleTick() + require.Same(t, cur, rig.exec.current) + } + rig.broadcastCompletion(session, reference.bankhash) + rig.drive(func() bool { return rig.block != nil }, "completed child after skips") + rig.finalizeAndCompare(reference) +} diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go new file mode 100644 index 000000000..d716e84a3 --- /dev/null +++ b/pkg/replay/streaming_test.go @@ -0,0 +1,1197 @@ +package replay + +import ( + "context" + "errors" + "sort" + "testing" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/blockstream" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// fakeStreamFeed stands in for the block source's view of the turbine feed. +type fakeStreamFeed struct { + events chan turbine.StreamEvent + status map[turbine.StreamGeneration]turbine.StreamStatus + pending map[turbine.StreamGeneration][]*turbine.StreamBatch + prioritized []uint64 +} + +func newFakeStreamFeed() *fakeStreamFeed { + return &fakeStreamFeed{ + events: make(chan turbine.StreamEvent, 64), + status: make(map[turbine.StreamGeneration]turbine.StreamStatus), + pending: make(map[turbine.StreamGeneration][]*turbine.StreamBatch), + } +} + +func (f *fakeStreamFeed) StreamEvents() <-chan turbine.StreamEvent { return f.events } + +func (f *fakeStreamFeed) StreamStatusOf(g turbine.StreamGeneration) turbine.StreamStatus { + if status, ok := f.status[g]; ok { + return status + } + return turbine.StreamGone +} + +func (f *fakeStreamFeed) PendingStreamBatches(g turbine.StreamGeneration, fromStart uint32) []*turbine.StreamBatch { + var out []*turbine.StreamBatch + for _, batch := range f.pending[g] { + if batch.Start >= fromStart { + out = append(out, batch) + } + } + sort.Slice(out, func(i, j int) bool { return out[i].Start < out[j].Start }) + return out +} + +func (f *fakeStreamFeed) PrioritizeStreamRepair(g turbine.StreamGeneration) { + f.prioritized = append(f.prioritized, g.Slot()) +} + +// fakeUnrootedState satisfies the tail interface for eligibility checks; no +// method is ever called on it. +type fakeUnrootedState struct{ unrootedState } + +// verifiedIdentities runs the real batch verifier so the batches carry the +// identities the assembler's verifier would attach. +func verifiedIdentities(t *testing.T, txs []*solana.Transaction) []txverify.VerifiedMessageIdentity { + t.Helper() + var verifier txverify.BatchVerifier + errs := make([]error, len(txs)) + identities := make([]txverify.VerifiedMessageIdentity, len(txs)) + verifier.VerifyWithMessageIdentities(txs, errs, identities) + for i, err := range errs { + require.NoError(t, err, "fixture transaction %d must verify", i) + } + return identities +} + +// streamingTestHarness is an executor with a stream already open on the group +// execution environment's bank (slot 42 on parent 41), which is what +// openStream would have produced without the bank machinery. +type streamingTestHarness struct { + env *groupExecutionEnv + feed *fakeStreamFeed + exec *streamingExecutor + gen turbine.StreamGeneration + parentID solana.Hash + frontier uint64 + lastCtx *sealevel.SlotCtx +} + +func newStreamingTestHarness(t *testing.T) *streamingTestHarness { + t.Helper() + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + t.Cleanup(env.cleanup) + feed := newFakeStreamFeed() + h := &streamingTestHarness{env: env, feed: feed, parentID: solana.Hash{7, 7, 7}, frontier: 41} + h.gen = turbine.NewDetachedStreamGeneration(env.exec.block.Slot) + feed.status[h.gen] = turbine.StreamActive + h.lastCtx = &sealevel.SlotCtx{Slot: 41, Epoch: env.exec.block.Epoch} + epochSchedule := sealevel.SysvarEpochSchedule{SlotsPerEpoch: 432000, LeaderScheduleSlotOffset: 432000, FirstNormalEpoch: 0, FirstNormalSlot: 0} + h.exec = newStreamingExecutor(streamingDeps{ + feed: feed, + epochSchedule: &epochSchedule, + tail: fakeUnrootedState{}, + transactionStatuses: NewTransactionStatusCache(), + alpenglowMode: true, + unrootedTailUsed: true, + lastSlotCtx: func() *sealevel.SlotCtx { return h.lastCtx }, + frontier: func() uint64 { return h.frontier }, + currentFeatures: func() *features.Features { return env.exec.block.Features }, + currentEpoch: func() uint64 { return env.exec.block.Epoch }, + rewardsInFlight: func() bool { return false }, + switchPending: func() bool { return false }, + executedBlockID: func(slot uint64) (solana.Hash, bool) { + if slot == 41 { + return h.parentID, true + } + return solana.Hash{}, false + }, + }) + env.exec.parentBankSysvars = &sealevel.BankSysvars{} + h.open() + return h +} + +// open installs the stream the way openStream does after its bank opened. +func (h *streamingTestHarness) open() { + h.exec.current = &streamingSlot{ + slot: h.env.exec.block.Slot, + generation: h.gen, + parentSlot: 41, + parentID: h.parentID, + exec: h.env.exec, + pending: make(map[uint32]*turbine.StreamBatch), + headerAt: time.Now(), + openedAt: time.Now(), + } + h.exec.handleEvent(h.event(h.header())) +} + +func (h *streamingTestHarness) header() *turbine.StreamBatch { + return turbine.NewDetachedStreamMarker(h.gen, 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) +} + +func (h *streamingTestHarness) batch(t *testing.T, start, end uint32, txs []*solana.Transaction) *turbine.StreamBatch { + t.Helper() + return turbine.NewDetachedStreamBatch(h.gen, start, end, txs, verifiedIdentities(t, txs)) +} + +func (h *streamingTestHarness) event(batch *turbine.StreamBatch) turbine.StreamEvent { + return turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: batch.Slot, Generation: batch.Generation, Batch: batch} +} + +// executed returns the block objects the stream has executed (in order); the +// bank itself ran stream-owned copies, which are checked to be copies of +// exactly those objects. +func (h *streamingTestHarness) executed() []*solana.Transaction { + if h.exec.current == nil { + return nil + } + return h.exec.current.origin +} + +// sameCopies asserts that executed holds fresh copies of want, in order: the +// same signatures and static keys, never the same objects, and never sharing +// the originals' account-key storage. +func sameCopies(t *testing.T, want, executed []*solana.Transaction) { + t.Helper() + require.Len(t, executed, len(want)) + for i := range want { + require.NotSame(t, want[i], executed[i], "transaction %d must be a copy", i) + require.Equal(t, want[i].Signatures, executed[i].Signatures, "transaction %d signatures", i) + require.Equal(t, want[i].Message.GetVersion(), executed[i].Message.GetVersion()) + require.Equal(t, want[i].Message.RecentBlockhash, executed[i].Message.RecentBlockhash) + if want[i].Message.GetVersion() == solana.MessageVersionV0 { + require.False(t, want[i].Message.IsResolved(), "transaction %d: the block's object must stay untouched", i) + } + if len(want[i].Message.AccountKeys) > 0 { + require.NotSame(t, &want[i].Message.AccountKeys[0], &executed[i].Message.AccountKeys[0], "transaction %d shares account-key storage", i) + } + } +} + +func (h *streamingTestHarness) discardReason() string { + return metrics.GlobalBlockReplay.StreamingExecution.DiscardReason +} + +func sameTransactions(t *testing.T, want, got []*solana.Transaction) { + t.Helper() + require.Len(t, got, len(want)) + for i := range want { + require.Same(t, want[i], got[i], "transaction %d", i) + } +} + +func TestStreamingConsumeExecutesContiguousGroupsInOrder(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 9, 500) + a, bb, c := h.batch(t, 1, 3, txs[0:3]), h.batch(t, 4, 6, txs[3:6]), h.batch(t, 7, 9, txs[6:9]) + + require.Equal(t, uint32(1), h.exec.current.nextStart, "header consumed") + h.exec.handleEvent(h.event(bb)) + require.Empty(t, h.executed(), "a gap before the batch holds it") + h.exec.handleEvent(h.event(a)) + sameTransactions(t, txs[0:6], h.executed()) + sameCopies(t, txs[0:6], h.env.exec.transactions) + require.Len(t, h.exec.current.groups, 1, "contiguous batches execute as one group") + require.Equal(t, uint32(7), h.exec.current.nextStart) + + // Duplicate wake-ups for consumed or pending ranges are ignored. + h.exec.handleEvent(h.event(a)) + h.exec.handleEvent(h.event(bb)) + sameTransactions(t, txs[0:6], h.executed()) + + h.exec.handleEvent(h.event(c)) + sameTransactions(t, txs, h.executed()) + sameCopies(t, txs, h.env.exec.transactions) + require.Len(t, h.exec.current.groups, 2) + require.Equal(t, uint32(10), h.exec.current.nextStart) + require.Equal(t, uint64(9), h.env.exec.processedTxCount) +} + +func TestStreamingTickPullsBatchesMissedByDroppedWakeups(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, MaxOpenAge: time.Minute} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 6, 600) + h.feed.pending[h.gen] = []*turbine.StreamBatch{h.batch(t, 4, 6, txs[3:6]), h.batch(t, 1, 3, txs[0:3])} + h.exec.handleTick() + sameTransactions(t, txs, h.executed()) + require.Len(t, h.exec.current.groups, 1) +} + +func TestStreamingConsumeHoldsUntilMinGroupBatches(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, MinGroupBatches: 2} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 9, 700) + a, bb, c := h.batch(t, 1, 3, txs[0:3]), h.batch(t, 4, 6, txs[3:6]), h.batch(t, 7, 9, txs[6:9]) + + h.exec.handleEvent(h.event(a)) + require.Empty(t, h.executed(), "one batch is below the group minimum") + require.Equal(t, uint32(1), h.exec.current.nextStart, "held batch is put back") + h.exec.handleEvent(h.event(bb)) + sameTransactions(t, txs[0:6], h.executed()) + + h.exec.handleEvent(h.event(c)) + require.Len(t, h.executed(), 6, "a lone trailing batch waits") + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamCompleted, Slot: c.Slot, Generation: h.gen}) + require.True(t, h.exec.current.completed) + sameTransactions(t, txs, h.executed()) +} + +func TestStreamingDiscardsOnUpdateParentDecodeErrorAndUnverified(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 3, 800) + h.exec.handleEvent(h.event(h.batch(t, 1, 3, txs))) + sameTransactions(t, txs, h.executed()) + update := turbine.NewDetachedStreamMarker(h.gen, 4, 4, turbine.StreamMarkerUpdateParent, 40, solana.Hash{1}) + h.exec.handleEvent(h.event(update)) + require.Nil(t, h.exec.current) + require.True(t, h.env.exec.closed, "discard closes the execution") + require.Equal(t, "update_parent", h.discardReason()) + require.Equal(t, uint64(1), metrics.GlobalBlockReplay.StreamingExecution.Discarded) + + h = newStreamingTestHarness(t) + broken := h.batch(t, 1, 3, txs) + broken.Err = errors.New("bad entry") + h.exec.handleEvent(h.event(broken)) + require.Nil(t, h.exec.current) + require.Equal(t, "decode_error", h.discardReason()) + + h = newStreamingTestHarness(t) + unverified := turbine.NewDetachedStreamBatch(h.gen, 1, 3, txs, nil) + _, verified, err := unverified.WaitVerification(context.Background()) + require.NoError(t, err) + require.False(t, verified) + h.exec.handleEvent(h.event(unverified)) + require.Nil(t, h.exec.current, "a batch the verifier did not admit is never self-verified") + require.Equal(t, "unverified_batch", h.discardReason()) + require.Empty(t, h.executed()) +} + +func TestStreamingGroupFailureDiscards(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 4, 900) + h.exec.executeFn = func(exec *blockExecution, group []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerify bool) error { + require.False(t, shouldVerify, "verified batches never re-verify") + return &DuplicateTransactionMessagesError{Slot: exec.block.Slot, DuplicateCount: 1} + } + h.exec.handleEvent(h.event(h.batch(t, 1, 2, txs[:2]))) + require.Nil(t, h.exec.current) + require.Equal(t, "duplicate_message", h.discardReason()) +} + +func TestStreamingIgnoresOtherGenerationsAndHonoursCancellation(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 3, 1000) + other := turbine.NewDetachedStreamGeneration(h.env.exec.block.Slot) + foreign := turbine.NewDetachedStreamBatch(other, 1, 3, txs, verifiedIdentities(t, txs)) + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: foreign.Slot, Generation: other, Batch: foreign}) + require.Empty(t, h.executed(), "another generation's batches are not this stream's") + require.NotNil(t, h.exec.current) + + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamCancelled, Slot: foreign.Slot, Generation: other, Reason: "reset"}) + require.NotNil(t, h.exec.current, "another generation's cancellation is ignored") + + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamCancelled, Slot: h.env.exec.block.Slot, Generation: h.gen, Reason: "reset"}) + require.Nil(t, h.exec.current) + require.Equal(t, "cancelled:reset", h.discardReason()) +} + +func TestStreamingTickEnforcesStatusAndAge(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, MaxOpenAge: 50 * time.Millisecond} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + + h := newStreamingTestHarness(t) + h.feed.status[h.gen] = turbine.StreamGone + h.exec.handleTick() + require.Nil(t, h.exec.current) + require.Equal(t, "gone", h.discardReason()) + + h = newStreamingTestHarness(t) + h.exec.current.openedAt = time.Now().Add(-time.Second) + h.exec.handleTick() + require.Nil(t, h.exec.current) + require.Equal(t, "timeout", h.discardReason()) + + h = newStreamingTestHarness(t) + h.feed.status[h.gen] = turbine.StreamDone + h.exec.current.openedAt = time.Now().Add(-100 * time.Millisecond) + h.exec.handleTick() + require.NotNil(t, h.exec.current, "a completed stream outlives MaxOpenAge while its block is emitted") + require.True(t, h.exec.current.completed) + h.exec.current.openedAt = time.Now().Add(-time.Minute) + h.exec.handleTick() + require.Nil(t, h.exec.current, "but not the hard cap") + require.Equal(t, "timeout", h.discardReason()) +} + +func TestStreamingDiscardUndoesGlobalSideEffects(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + slot := h.env.exec.block.Slot + slotCtx := h.env.exec.slotCtx + slotCtx.DeferVoteCachePublication = true + voteKey := solana.PublicKey{9} + putVoteCacheItem(slotCtx, voteKey, &sealevel.VoteStateVersions{}) + markSlotVoteStakeDirty(slotCtx) + require.Nil(t, global.VoteCacheItem(voteKey), "deferred put stays off the global cache") + before := len(global.PendingStakeEntriesSnapshot()) + global.EnqueuePendingStakePubkey(slot, solana.PublicKey{8}) + require.Len(t, global.PendingStakeEntriesSnapshot(), before+1) + + h.exec.discard("test") + require.Nil(t, slotCtx.PendingVoteCache) + require.False(t, slotCtx.VoteStakeDirty) + require.Nil(t, global.VoteCacheItem(voteKey)) + require.Len(t, global.PendingStakeEntriesSnapshot(), before, "the stream's stake index entries are dropped") + require.False(t, h.exec.matches(slot)) + require.Nil(t, h.exec.tick(), "no poll timer while idle") +} + +func TestStreamingEligibility(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + h.exec.discard("reset for eligibility") + header := h.header() + require.Equal(t, "", h.exec.eligibility(header)) + + cases := []struct { + name string + mutate func() + want string + }{ + {"generation gone", func() { h.feed.status[h.gen] = turbine.StreamGone }, "generation no longer active"}, + {"generation done", func() { h.feed.status[h.gen] = turbine.StreamDone }, "generation no longer active"}, + {"no parent context", func() { h.lastCtx = nil }, "no executed parent context"}, + {"frontier moved", func() { h.frontier = 42 }, "slot 42 on parent 41 does not extend the executed frontier 42 (parent context 41)"}, + {"parent is not the executed slot", func() { h.lastCtx = &sealevel.SlotCtx{Slot: 40} }, "slot 42 on parent 41 does not extend the executed frontier 41 (parent context 40)"}, + {"parent id mismatch", func() { h.parentID = solana.Hash{1} }, "parent block id does not match the executed parent"}, + {"switch pending", func() { h.exec.deps.switchPending = func() bool { return true } }, "fork switch pending"}, + {"epoch boundary", func() { h.exec.deps.currentEpoch = func() uint64 { return 99 } }, "epoch boundary"}, + {"rewards", func() { h.exec.deps.rewardsInFlight = func() bool { return true } }, "partitioned rewards in flight"}, + {"no features", func() { h.exec.deps.currentFeatures = func() *features.Features { return nil } }, "no feature set"}, + {"no tail", func() { h.exec.deps.tail = nil }, "requires alpenglow rooted-durable replay"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + saved := *h + savedDeps := h.exec.deps + savedStatus := h.feed.status[h.gen] + tc.mutate() + require.Equal(t, tc.want, h.exec.eligibility(header)) + *h = saved + h.exec.deps = savedDeps + h.feed.status[h.gen] = savedStatus + }) + } +} + +func TestStreamingRememberHeaderAndTryOpenBounds(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + h.exec.discard("idle") + + stale := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(41), 0, 0, turbine.StreamMarkerHeader, 40, solana.Hash{}) + h.exec.handleEvent(h.event(stale)) + require.Empty(t, h.exec.headers, "headers at or below the frontier are not kept") + + ahead := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(44), 0, 0, turbine.StreamMarkerHeader, 43, solana.Hash{}) + h.exec.handleEvent(h.event(ahead)) + require.Contains(t, h.exec.headers, uint64(44), "a header ahead of the frontier waits for its parent") + require.Nil(t, h.exec.current) + + // The next slot's header is ineligible (its generation is unknown to the + // feed), so it is dropped rather than opened; nothing else changes. + next := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(42), 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) + h.exec.handleEvent(h.event(next)) + require.Nil(t, h.exec.current) + require.NotContains(t, h.exec.headers, uint64(42)) + require.Contains(t, h.exec.headers, uint64(44)) + + h.frontier = 44 + h.exec.handleTick() + require.Empty(t, h.exec.headers, "advancing the frontier past a remembered header drops it") + + StreamingExecutionCfg.Enabled = false + h.frontier = 41 + h.exec.headers[42] = next + h.exec.tryOpen() + require.Contains(t, h.exec.headers, uint64(42), "disabled: nothing opens") +} + +// A dropped header wake-up is recovered from the assembler by the next +// wake-up for the generation (the pending list is authoritative after a +// drop); the usual open path follows. The lookup is only made for the slot +// within the bounded lookahead, never for a generation this executor retired, and only finds +// a header that is decoded. +func TestStreamingRecoversHeaderFromPendingAfterDroppedWakeup(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 3, 700) + g := turbine.NewDetachedStreamGeneration(42) + header := turbine.NewDetachedStreamMarker(g, 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) + batch := turbine.NewDetachedStreamBatch(g, 1, 3, txs, verifiedIdentities(t, txs)) + h.feed.pending[g] = []*turbine.StreamBatch{batch, header} + + h.exec.recoverHeader(42, g) + require.Empty(t, h.exec.headers, "the open stream's own slot is not looked up (its successor is; see below)") + + h.exec.discard("idle") + require.Equal(t, h.gen, h.exec.retired[42], "the discarded generation is retired") + h.exec.recoverHeader(42, g) + require.Same(t, header, h.exec.headers[42], "the batch wake-up recovered the decoded header") + + // Declined once (the feed does not know the generation, so it is + // ineligible), the generation is retired and no later wake-up brings it + // back; whole-block execution owns the slot. + h.exec.tryOpen() + require.Nil(t, h.exec.current) + require.Empty(t, h.exec.headers) + require.Equal(t, g, h.exec.retired[42]) + h.exec.recoverHeader(42, g) + require.Empty(t, h.exec.headers, "a retired generation is never recovered") + + // A new generation of the slot (after a reset) is recoverable again, but + // only once its header is decoded. + renewed := turbine.NewDetachedStreamGeneration(42) + renewedHeader := turbine.NewDetachedStreamMarker(renewed, 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) + h.feed.pending[renewed] = []*turbine.StreamBatch{turbine.NewDetachedStreamBatch(renewed, 1, 3, txs, verifiedIdentities(t, txs))} + h.exec.recoverHeader(42, renewed) + require.Empty(t, h.exec.headers, "a header that is not decoded yet cannot be recovered") + h.feed.pending[renewed] = append(h.feed.pending[renewed], renewedHeader) + h.exec.recoverHeader(42, renewed) + require.Same(t, renewedHeader, h.exec.headers[42]) + delete(h.exec.headers, 42) + + ahead := turbine.NewDetachedStreamGeneration(44) + h.feed.pending[ahead] = []*turbine.StreamBatch{turbine.NewDetachedStreamMarker(ahead, 0, 0, turbine.StreamMarkerHeader, 43, solana.Hash{})} + h.exec.recoverHeader(44, ahead) + require.Contains(t, h.exec.headers, uint64(44), "nearby header can be recovered before its parent is ready") + delete(h.exec.headers, 44) + + h.exec.recoverHeader(42, turbine.StreamGeneration{}) + require.Empty(t, h.exec.headers, "a zero generation has nothing pending") + + // While a stream is open, its successor is the slot that could open next. + h.frontier = 42 + h.exec.pruneHeaders(h.frontier) + require.Empty(t, h.exec.retired, "retirements at or below the frontier are pruned") + h.frontier = 41 + h.open() + successor := turbine.NewDetachedStreamGeneration(43) + successorHeader := turbine.NewDetachedStreamMarker(successor, 0, 0, turbine.StreamMarkerHeader, 42, solana.Hash{1}) + h.feed.pending[successor] = []*turbine.StreamBatch{successorHeader} + h.exec.recoverHeader(43, successor) + require.Same(t, successorHeader, h.exec.headers[43], "the open stream's successor is recoverable") + require.NotNil(t, h.exec.current, "the open stream is untouched") +} + +func (h *streamingTestHarness) matchingBlock(t *testing.T, txs []*solana.Transaction) *b.Block { + t.Helper() + shell := h.env.exec.block + block := &b.Block{ + Slot: shell.Slot, + Epoch: shell.Epoch, + ParentSlot: shell.ParentSlot, + ParentBankhash: shell.ParentBankhash, + Features: shell.Features, + FromLiveStream: true, + SourceParentSlot: 41, + AlpenglowParentBlockID: h.parentID, + HasAlpenglowParentBlockID: true, + Transactions: txs, + } + return block +} + +func TestStreamingHandshake(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 4, 1100) + h.exec.handleEvent(h.event(h.batch(t, 1, 3, txs[:3]))) + sameTransactions(t, txs[:3], h.executed()) + sysvars := h.env.exec.parentBankSysvars + + require.Equal(t, "", h.exec.handshake(h.matchingBlock(t, txs), sysvars), "the block extends the executed prefix") + require.Equal(t, "", h.exec.handshake(h.matchingBlock(t, txs[:3]), sysvars), "the block may end exactly at the prefix") + + cases := []struct { + name string + mutate func(block *b.Block) + want string + }{ + {"slot", func(block *b.Block) { block.Slot++ }, "slot"}, + {"skipped", func(block *b.Block) { block.IsSkipped = true }, "not a live block"}, + {"rpc block", func(block *b.Block) { block.FromLiveStream = false }, "not a live block"}, + {"parent id", func(block *b.Block) { block.AlpenglowParentBlockID = [32]byte{1} }, "parent"}, + {"parent slot", func(block *b.Block) { block.SourceParentSlot = 40 }, "parent"}, + {"configured parent", func(block *b.Block) { block.ParentBankhash = [32]byte{2} }, "configured parent"}, + {"features", func(block *b.Block) { block.Features = features.NewFeaturesDefault() }, "features"}, + {"epoch", func(block *b.Block) { block.Epoch++ }, "epoch"}, + {"epoch accounts", func(block *b.Block) { block.EpochUpdatedAccts = append(block.EpochUpdatedAccts, nil) }, "epoch account updates"}, + {"shorter", func(block *b.Block) { block.Transactions = block.Transactions[:2] }, "shorter than executed prefix"}, + {"different transaction", func(block *b.Block) { + replacement := transferTransactions(t, 1, 1101)[0] // same bytes as txs[1], different object + block.Transactions[1] = replacement + }, "transaction 1 identity"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + block := h.matchingBlock(t, append([]*solana.Transaction(nil), txs...)) + tc.mutate(block) + require.Equal(t, tc.want, h.exec.handshake(block, sysvars)) + }) + } + require.Equal(t, "parent sysvars", h.exec.handshake(h.matchingBlock(t, txs), &sealevel.BankSysvars{})) + require.Equal(t, "parent sysvars", h.exec.handshake(h.matchingBlock(t, txs), nil)) + h.feed.status[h.gen] = turbine.StreamGone + require.Equal(t, "generation gone", h.exec.handshake(h.matchingBlock(t, txs), sysvars)) +} + +func TestStreamingFinalizeFallsBackOnMismatch(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 3, 1200) + h.exec.handleEvent(h.event(h.batch(t, 1, 3, txs))) + block := h.matchingBlock(t, txs) + block.Transactions[0] = transferTransactions(t, 1, 1200)[0] + + slotCtx, ok, err := h.exec.finalize(block, h.env.exec.parentBankSysvars) + require.NoError(t, err) + require.False(t, ok, "the caller must execute the block whole") + require.Nil(t, slotCtx) + require.Nil(t, h.exec.current) + require.Equal(t, "prefix_mismatch:transaction 0 identity", h.discardReason()) + require.True(t, h.env.exec.closed) + + _, ok, err = h.exec.finalize(block, nil) + require.NoError(t, err) + require.False(t, ok, "no stream, nothing to finalize") + require.False(t, h.exec.matches(block.Slot)) +} + +// finalizeFailureHarness prepares a stream whose executed prefix matches the +// block (three of four transfers executed) and instruments every undo hook, +// so a post-handshake failure can be checked for the discard contract. +type finalizeFailureHarness struct { + *streamingTestHarness + txs []*solana.Transaction + restores int + voteKey solana.PublicKey + stakeBefore int + openedPending int +} + +func newFinalizeFailureHarness(t *testing.T) *finalizeFailureHarness { + t.Helper() + h := &finalizeFailureHarness{streamingTestHarness: newStreamingTestHarness(t), voteKey: solana.PublicKey{3, 3, 3}} + h.txs = transferTransactions(t, 4, 1300) + h.exec.handleEvent(h.event(h.batch(t, 1, 3, h.txs[:3]))) + sameTransactions(t, h.txs[:3], h.executed()) + h.exec.current.restoreSysvarCache = func() { h.restores++ } + slotCtx := h.env.exec.slotCtx + slotCtx.DeferVoteCachePublication = true + slotCtx.TrackProgramCacheAdds = true + putVoteCacheItem(slotCtx, h.voteKey, &sealevel.VoteStateVersions{}) + slotCtx.RecordProgramCacheAdd(solana.PublicKey{4}) + h.stakeBefore = len(global.PendingStakeEntriesSnapshot()) + global.EnqueuePendingStakePubkey(h.env.exec.block.Slot, solana.PublicKey{5}) + return h +} + +func (h *finalizeFailureHarness) assertUndone(t *testing.T, reason string) { + t.Helper() + require.Nil(t, h.exec.current, "the stream no longer owns a bank") + require.True(t, h.env.exec.closed, "the execution is closed") + require.Equal(t, reason, h.discardReason()) + require.Equal(t, 1, h.restores, "the legacy sysvar cache is restored once") + require.Nil(t, global.VoteCacheItem(h.voteKey), "unpublished vote-cache entries never reach the global cache") + require.Nil(t, h.env.exec.slotCtx.PendingVoteCache) + require.Empty(t, h.env.exec.slotCtx.TakeProgramCacheAdds(), "tracked program-cache adds were consumed by the undo") + require.Len(t, global.PendingStakeEntriesSnapshot(), h.stakeBefore, "the slot's stake index entries are dropped") + require.Nil(t, h.exec.tick()) +} + +func TestStreamingFinalizeSuffixFailureUndoesTheStream(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newFinalizeFailureHarness(t) + suffixErr := errors.New("suffix exploded") + calls := 0 + h.exec.executeFn = func(exec *blockExecution, group []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerify bool) error { + calls++ + sameTransactions(t, h.txs[3:], group) + require.Equal(t, 1, identities.Len()) + require.True(t, shouldVerify, "an unmarked block's suffix is verified like any whole block") + return suffixErr + } + block := h.matchingBlock(t, h.txs) + + slotCtx, ok, err := h.exec.finalize(block, h.env.exec.parentBankSysvars) + require.True(t, ok, "the handshake passed: the failure is the block's") + require.Nil(t, slotCtx) + require.ErrorIs(t, err, suffixErr) + var final *streamingFinalizeError + require.ErrorAs(t, err, &final) + require.Equal(t, 1, calls) + h.assertUndone(t, "finalize:suffix") +} + +func TestStreamingFinalizeFooterRejectionUndoesTheStream(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newFinalizeFailureHarness(t) + // An Alpenglow bank requires the footer before it executes the suffix; a + // live block without one is rejected exactly as ProcessBlock rejects it. + h.exec.deps.alpenglowClock = true + h.env.exec.block.Features.EnableFeature(features.AlpenglowDevContext, 0) + h.exec.executeFn = func(*blockExecution, []*solana.Transaction, *b.PreparedTransactionMessageIdentities, bool) error { + t.Fatal("the suffix must not execute after a footer rejection") + return nil + } + block := h.matchingBlock(t, h.txs) + require.False(t, block.HasAlpenglowFooter) + + slotCtx, ok, err := h.exec.finalize(block, h.env.exec.parentBankSysvars) + require.True(t, ok) + require.Nil(t, slotCtx) + require.ErrorContains(t, err, "missing block footer") + h.assertUndone(t, "finalize:footer_clock") +} + +func TestStreamingFinalizeProcessedCountMismatchUndoesTheStream(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newFinalizeFailureHarness(t) + // A suffix that "succeeds" without recording its transactions leaves the + // executed counts short of the whole-block plan. + h.exec.executeFn = func(*blockExecution, []*solana.Transaction, *b.PreparedTransactionMessageIdentities, bool) error { + return nil + } + block := h.matchingBlock(t, h.txs) + + _, ok, err := h.exec.finalize(block, h.env.exec.parentBankSysvars) + require.True(t, ok) + require.ErrorContains(t, err, "processed 3 transactions") + h.assertUndone(t, "finalize:counts") +} + +// recordingStreamer is the wait loop's view of an executor. +type recordingStreamer struct { + ch chan turbine.StreamEvent + tickCh chan time.Time + handled []turbine.StreamEvent + ticks int +} + +func (r *recordingStreamer) events() <-chan turbine.StreamEvent { return r.ch } +func (r *recordingStreamer) tick() <-chan time.Time { return r.tickCh } +func (r *recordingStreamer) handleEvent(e turbine.StreamEvent) { r.handled = append(r.handled, e) } +func (r *recordingStreamer) handleTick() { r.ticks++ } + +func TestWaitForReplayInputDispatchesStreamWakeups(t *testing.T) { + streamer := &recordingStreamer{ch: make(chan turbine.StreamEvent, 4), tickCh: make(chan time.Time, 4)} + block := &b.Block{Slot: 5} + script := []blockstream.ReplayInput{ + {StreamEvent: &turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: 6}}, + {StreamTick: true}, + {DecisionChanged: true}, + {StreamEvent: &turbine.StreamEvent{Kind: turbine.StreamCompleted, Slot: 6}}, + {Block: block}, + } + var seenEvents []<-chan turbine.StreamEvent + var seenTicks []<-chan time.Time + next := func(ctx context.Context, decisionChanges <-chan struct{}, events <-chan turbine.StreamEvent, tick <-chan time.Time) blockstream.ReplayInput { + seenEvents = append(seenEvents, events) + seenTicks = append(seenTicks, tick) + in := script[0] + script = script[1:] + return in + } + sweeps := 0 + sweep := func() *CertifiedSwitch { sweeps++; return nil } + got, parentSwitch, sw := waitForReplayInput(context.Background(), next, sweep, make(chan struct{}), time.Second, streamer) + require.Same(t, block, got) + require.Nil(t, parentSwitch) + require.Nil(t, sw) + require.Len(t, streamer.handled, 2, "feed wake-ups are handled and never end the wait") + require.Equal(t, turbine.StreamCompleted, streamer.handled[1].Kind) + require.Equal(t, 2, streamer.ticks, "one tick on entry, one from the timer") + require.Equal(t, 5, sweeps, "every wait is preceded by a sweep") + for _, ch := range seenEvents { + require.Equal(t, (<-chan turbine.StreamEvent)(streamer.ch), ch) + } + for _, ch := range seenTicks { + require.Equal(t, (<-chan time.Time)(streamer.tickCh), ch) + } +} + +func TestWaitForReplayInputStreamerWithoutSweepBlocksWithoutPolling(t *testing.T) { + streamer := &recordingStreamer{ch: make(chan turbine.StreamEvent, 1)} + calls := 0 + next := func(ctx context.Context, decisionChanges <-chan struct{}, events <-chan turbine.StreamEvent, tick <-chan time.Time) blockstream.ReplayInput { + calls++ + require.Nil(t, decisionChanges, "decision wake-ups stay disabled without a sweep") + _, hasDeadline := ctx.Deadline() + require.False(t, hasDeadline, "no poll timeout without a sweep") + if calls == 1 { + return blockstream.ReplayInput{StreamTick: true} + } + return blockstream.ReplayInput{} + } + block, parentSwitch, sw := waitForReplayInput(context.Background(), next, nil, make(chan struct{}), time.Second, streamer) + require.Nil(t, block) + require.Nil(t, parentSwitch) + require.Nil(t, sw) + require.Equal(t, 2, calls) + require.Equal(t, 2, streamer.ticks) +} + +func TestWaitForReplayInputWithoutStreamerIsUnchanged(t *testing.T) { + block := &b.Block{Slot: 9} + next := func(ctx context.Context, decisionChanges <-chan struct{}, events <-chan turbine.StreamEvent, tick <-chan time.Time) blockstream.ReplayInput { + require.Nil(t, events) + require.Nil(t, tick) + require.Nil(t, decisionChanges) + return blockstream.ReplayInput{Block: block} + } + got, _, sw := waitForReplayInput(context.Background(), next, nil, make(chan struct{}), time.Second, nil) + require.Same(t, block, got) + require.Nil(t, sw) + + var nilStreamer *streamingExecutor + require.Nil(t, nilStreamer.tick()) + require.Nil(t, nilStreamer.events()) + require.False(t, nilStreamer.matches(1)) + nilStreamer.handleTick() + nilStreamer.discard("noop") + nilStreamer.discardSlot(1, "noop") + nilStreamer.shutdown() + _, ok, err := nilStreamer.finalize(block, nil) + require.False(t, ok) + require.NoError(t, err) +} + +func TestStreamingConfigDefaults(t *testing.T) { + var cfg StreamingExecutionConfig + require.Equal(t, defaultStreamingWorkers, cfg.workers(0)) + require.Equal(t, 2, cfg.workers(2)) + require.Equal(t, defaultStreamingWorkers, cfg.workers(64)) + cfg.Workers = 8 + require.Equal(t, 8, cfg.workers(0)) + require.Equal(t, 3, cfg.workers(3)) + require.Equal(t, defaultStreamingMaxAge, cfg.maxOpenAge()) + cfg.MaxOpenAge = time.Second + require.Equal(t, time.Second, cfg.maxOpenAge()) +} + +func TestStreamingGapSelectionAndSafety(t *testing.T) { + h := newStreamingTestHarness(t) + defer h.exec.shutdown() + makeHeader := func(slot, parent uint64, id solana.Hash) *turbine.StreamBatch { + g := turbine.NewDetachedStreamGeneration(slot) + h.feed.status[g] = turbine.StreamActive + return turbine.NewDetachedStreamMarker(g, 0, 0, turbine.StreamMarkerHeader, parent, id) + } + gap := makeHeader(46, 41, h.parentID) + h.exec.headers[46] = gap + h.exec.headers[48] = makeHeader(48, 41, h.parentID) + h.exec.headers[44] = makeHeader(44, 43, h.parentID) + require.Same(t, gap, h.exec.nextHeader(h.frontier), "earliest direct child, not a grandchild") + require.Empty(t, h.exec.eligibility(gap)) + require.NotEmpty(t, h.exec.eligibility(makeHeader(46, 41, solana.Hash{99}))) + require.NotEmpty(t, h.exec.eligibility(makeHeader(46, 40, h.parentID))) + require.NotEmpty(t, h.exec.eligibility(makeHeader(74, 41, h.parentID)), "lookahead is bounded") + h.exec.deps.switchPending = func() bool { return true } + require.Equal(t, "fork switch pending", h.exec.eligibility(gap)) + h.exec.deps.switchPending = func() bool { return false } + h.exec.headers[42] = makeHeader(42, 41, h.parentID) + require.Equal(t, uint64(42), h.exec.nextHeader(h.frontier).Slot) + + // A late real bank in the unresolved gap must restore all speculative + // state before that bank is validated, configured or executed. + h.exec.current.slot = 46 + cur := h.exec.current + h.exec.beforeBlock(&b.Block{Slot: 42, IsSkipped: true}) + require.Same(t, cur, h.exec.current) + h.exec.beforeBlock(&b.Block{Slot: 42}) + require.Nil(t, h.exec.current) + require.True(t, cur.exec.closed) + require.Equal(t, h.gen, h.exec.retired[46]) +} + +// The open timeline attributes every millisecond between the header's decode +// and the open to exactly one of: the parent's arrival (header decoded before +// the parent's last shred), the parent's replay (last shred → replay result), +// or the loop (nothing left to wait for, still not opened). +func TestStreamingTimelineAttributesOpenDelay(t *testing.T) { + t0 := time.Unix(1_700_000_000, 0) + at := func(ms int) time.Time { return t0.Add(time.Duration(ms) * time.Millisecond) } + block := &b.Block{ShredFullNanos: at(300).UnixNano()} + record := func(cur *streamingSlot) metrics.StreamingExecution { + var out metrics.StreamingExecution + cur.recordTimeline(&out, block, at(320)) + return out + } + ms := func(timing metrics.Timing) float64 { return float64(timing.SumNanoseconds) / 1e6 } + + // Header decoded 50 ms before the parent's last shred, parent admitted + // 20 ms after that and replayed 10 ms later, opened 5 ms after; the + // header wake-up was handled 10 ms after decode. + cur := &streamingSlot{ + headerAt: at(0), headerSeenAt: at(10), openedAt: at(85), + parentFullNanos: at(50).UnixNano(), parentAdmittedAt: at(70), parentReplayedAt: at(80), + groups: []streamingGroup{{startedAt: at(90), finishedAt: at(120), transactions: 3}}, + } + r := record(cur) + require.Equal(t, at(0).UnixNano(), r.HeaderReadyNanos) + require.Equal(t, at(10).UnixNano(), r.HeaderSeenNanos) + require.Equal(t, at(50).UnixNano(), r.ParentFullNanos) + require.Equal(t, at(70).UnixNano(), r.ParentAdmittedNanos) + require.Equal(t, at(80).UnixNano(), r.ParentReplayedNanos) + require.Equal(t, at(85).UnixNano(), r.OpenedNanos) + require.Equal(t, at(90).UnixNano(), r.FirstGroupStartNanos) + require.Equal(t, at(300).UnixNano(), r.FullNanos) + require.Equal(t, at(320).UnixNano(), r.FinalizeStartNanos) + require.Equal(t, 50.0, ms(r.OpenWaitParentArrival)) + require.Equal(t, 30.0, ms(r.OpenWaitParentReplay)) + require.Equal(t, 20.0, ms(r.OpenWaitParentPreAdmission), "the parent sat in the source 20 ms after its last shred") + require.Equal(t, 10.0, ms(r.OpenWaitParentPostAdmission)) + require.Equal(t, 5.0, ms(r.OpenWaitLoop)) + require.Equal(t, uint64(1), r.OpenWaitLoop.Count) + + // The parent was admitted before the child's header was decoded (its + // queueing cannot have held the child): the replay wait is all execution. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(1), openedAt: at(40), parentFullNanos: at(-30).UnixNano(), parentAdmittedAt: at(-5), parentReplayedAt: at(30)} + r = record(cur) + require.Zero(t, r.OpenWaitParentPreAdmission.Count) + require.Equal(t, 30.0, ms(r.OpenWaitParentPostAdmission)) + require.Equal(t, 30.0, ms(r.OpenWaitParentReplay)) + + // Parent fully received before the header was even decoded: no arrival + // wait; the parent's replay wait starts at the header. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(1), openedAt: at(40), parentFullNanos: at(-20).UnixNano(), parentReplayedAt: at(30)} + r = record(cur) + require.Zero(t, r.OpenWaitParentArrival.Count) + require.Equal(t, 30.0, ms(r.OpenWaitParentReplay)) + require.Equal(t, 10.0, ms(r.OpenWaitLoop)) + + // Header handled only after the parent was replayed (the wake-up sat in + // the channel): include the queued wake-up before the header was seen. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(90), openedAt: at(95), parentFullNanos: at(20).UnixNano(), parentReplayedAt: at(60)} + r = record(cur) + require.Equal(t, 20.0, ms(r.OpenWaitParentArrival)) + require.Equal(t, 40.0, ms(r.OpenWaitParentReplay)) + require.Equal(t, 35.0, ms(r.OpenWaitLoop)) + require.Equal(t, 95.0, ms(r.OpenWaitParentArrival)+ms(r.OpenWaitParentReplay)+ms(r.OpenWaitLoop)) + + // The loop wait splits at the first wait entry after the parent: a header + // remembered while the parent executed waited through the parent's + // post-replay tail (replayed → wait entry) and then its dispatch (wait + // entry → opened). + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(5), openedAt: at(130), parentFullNanos: at(-100).UnixNano(), parentReplayedAt: at(60), waitEnteredAt: at(125)} + r = record(cur) + require.Equal(t, at(125).UnixNano(), r.WaitEnteredNanos) + require.Equal(t, 70.0, ms(r.OpenWaitLoop)) + require.Equal(t, 65.0, ms(r.OpenWaitPostReplay)) + require.Equal(t, 5.0, ms(r.OpenWaitDispatch)) + + // A header seen only after the wait entry (it arrived while the loop was + // already waiting) includes both the tail and queued header dispatch. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(140), openedAt: at(141), parentFullNanos: at(-100).UnixNano(), parentReplayedAt: at(60), waitEnteredAt: at(125)} + r = record(cur) + require.Equal(t, 65.0, ms(r.OpenWaitPostReplay)) + require.Equal(t, 16.0, ms(r.OpenWaitDispatch)) + require.Equal(t, 81.0, ms(r.OpenWaitLoop)) + require.Contains(t, cur.openTimeline(), "wait entered +125.0ms") + + // Unknown parent instants (no mark, or a re-based frontier): only the + // loop wait, from the header being seen. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(0), openedAt: at(70)} + r = record(cur) + require.Zero(t, r.ParentFullNanos) + require.Zero(t, r.ParentReplayedNanos) + require.Zero(t, r.OpenWaitParentArrival.Count) + require.Zero(t, r.OpenWaitParentReplay.Count) + require.Zero(t, r.OpenWaitParentPreAdmission.Count) + require.Zero(t, r.OpenWaitParentPostAdmission.Count) + require.Equal(t, 70.0, ms(r.OpenWaitLoop)) + require.Zero(t, r.OpenWaitPostReplay.Count) + require.Zero(t, r.OpenWaitDispatch.Count) + require.Zero(t, r.WaitEnteredNanos) + require.Zero(t, r.FirstGroupStartNanos, "no group ran") + require.Contains(t, cur.openTimeline(), "parent full ?") + require.Contains(t, cur.openTimeline(), "opened +70.0ms") +} + +// A block executed whole says why no stream opened for it, from the +// executor's per-slot observation of its header. +func TestStreamingNoteWholeBlockReasons(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + reset := func() { metrics.GlobalBlockReplay.StreamingExecution = metrics.StreamingExecution{} } + reason := func() string { return metrics.GlobalBlockReplay.StreamingExecution.NotOpenedReason } + + // Discarded stream, recorded on the observation (the record's own + // Discarded/DiscardReason may or may not have survived the loop's + // per-attempt reset; the observation always has it). + reset() + h.exec.observed[42] = &streamingObservation{generation: h.gen, parentSlot: 41, readyAt: time.Now(), seenAt: time.Now(), frontierAtSeen: 41} + h.exec.discard("timeout") + h.exec.noteWholeBlock(&b.Block{Slot: 42, ShredFullNanos: 123}) + require.Equal(t, "discarded:timeout", reason()) + require.Equal(t, int64(123), metrics.GlobalBlockReplay.StreamingExecution.FullNanos) + require.Positive(t, metrics.GlobalBlockReplay.StreamingExecution.HeaderReadyNanos) + require.Positive(t, metrics.GlobalBlockReplay.StreamingExecution.HeaderSeenNanos) + + // Never saw a header. + reset() + delete(h.exec.observed, 42) + h.exec.noteWholeBlock(&b.Block{Slot: 42}) + require.Equal(t, "header_not_seen", reason()) + + // Declined: the header's generation is unknown to the feed (gone). + reset() + declined := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(42), 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) + h.exec.handleEvent(h.event(declined)) + require.Nil(t, h.exec.current) + h.exec.noteWholeBlock(&b.Block{Slot: 42}) + require.Contains(t, reason(), "declined:generation no longer active") + + // Waiting: a header whose parent is not the executed frontier stays + // remembered, and a whole-block execution of it reports what it waited on. + reset() + waiting := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(44), 0, 0, turbine.StreamMarkerHeader, 43, solana.Hash{}) + h.exec.handleEvent(h.event(waiting)) + require.Contains(t, h.exec.headers, uint64(44)) + h.exec.noteWholeBlock(&b.Block{Slot: 44}) + require.Equal(t, "waiting_for_parent:header_on_parent_43_seen_at_frontier_41", reason()) + + // Skips never report; a nil executor is a no-op. + reset() + h.exec.noteWholeBlock(&b.Block{Slot: 44, IsSkipped: true}) + require.Empty(t, reason()) + var none *streamingExecutor + none.noteWholeBlock(&b.Block{Slot: 44}) + require.Empty(t, reason()) + + // Observations are pruned with the frontier. + h.frontier = 44 + h.exec.pruneHeaders(h.frontier) + require.Empty(t, h.exec.observed) +} + +// The per-group record splits verification waits and execution at the last +// shred, so the execution FullToReplayed paid for is visible separately +// from the work hidden behind reception. +func TestStreamingGroupRecordSplitsAtFull(t *testing.T) { + t0 := time.Unix(1_700_000_000, 0) + at := func(ms int) time.Time { return t0.Add(time.Duration(ms) * time.Millisecond) } + ms := func(timing metrics.Timing) float64 { return float64(timing.SumNanoseconds) / 1e6 } + cur := &streamingSlot{slot: 42, groups: []streamingGroup{ + {readyAt: at(90), joinedAt: at(92), startedAt: at(92), finishedAt: at(120), batches: 3, transactions: 900}, // entirely before full + {readyAt: at(250), joinedAt: at(310), startedAt: at(310), finishedAt: at(340), batches: 9, transactions: 5000}, // waited 60 ms for the verifier, 10 of them after full; ran after full + {readyAt: at(280), joinedAt: at(281), startedAt: at(281), finishedAt: at(320), batches: 1, transactions: 300}, // straddles full + {readyAt: at(340), joinedAt: at(340), startedAt: at(340), finishedAt: at(350), transactions: 40, suffix: true}, + }} + var r metrics.StreamingExecution + cur.recordGroups(&r, at(300)) + require.Equal(t, 63.0, ms(r.GroupJoinAssembly)) + require.Equal(t, uint64(3), r.GroupJoinAssembly.Count, "the suffix has no verifier wait") + require.Equal(t, 10.0, ms(r.GroupJoinAssemblyAfterFull)) + require.Equal(t, uint64(1), r.GroupJoinAssemblyAfterFull.Count) + require.Equal(t, 60.0, ms(r.TxLoopAfterFull), "30 + 20 + 10 ms of execution after the last shred") + require.Equal(t, uint64(3), r.TxLoopAfterFull.Count) + require.Equal(t, uint64(1), r.GroupsStraddlingFull) + require.Equal(t, uint64(5000), r.LargestGroupTransactions) + require.Equal(t, uint64(9), r.LargestGroupBatches) + require.Equal(t, at(350).UnixNano(), r.LastGroupEndNanos) + line := cur.groupTimeline(at(300), 3) + require.Contains(t, line, "[#0 b=3 tx=900 ready-210.0 joined-208.0 exec-208.0..-180.0]") + require.Contains(t, line, "…(1 more)") + require.Contains(t, line, "[#3 suffix b=0 tx=40 ready+40.0 joined+40.0 exec+40.0..+50.0]") + require.NotContains(t, line, "#2 ") + + // No full instant (a block that did not arrive as shreds): nothing is + // "after full", waits and sizes still count. + var whole metrics.StreamingExecution + cur.recordGroups(&whole, time.Time{}) + require.Equal(t, 63.0, ms(whole.GroupJoinAssembly)) + require.Zero(t, whole.GroupJoinAssemblyAfterFull.Count) + require.Zero(t, whole.TxLoopAfterFull.Count) + require.Zero(t, whole.GroupsStraddlingFull) + require.Equal(t, uint64(5000), whole.LargestGroupTransactions) + require.Equal(t, at(350).UnixNano(), whole.LastGroupEndNanos) +} + +// The group stages account for preparation even when it crosses completion. +func TestStreamingGroupPreparationAccounting(t *testing.T) { + base := time.Unix(1700000000, 0) + at := func(ms int) time.Time { return base.Add(time.Duration(ms) * time.Millisecond) } + cur := &streamingSlot{groups: []streamingGroup{{readyAt: at(0), joinedAt: at(10), startedAt: at(30), finishedAt: at(50)}}} + var r metrics.StreamingExecution + cur.recordGroups(&r, at(20)) + require.Equal(t, uint64(10*time.Millisecond), r.GroupJoinAssembly.SumNanoseconds) + require.Equal(t, uint64(20*time.Millisecond), r.GroupPreparation.SumNanoseconds) + require.Zero(t, r.GroupJoinAssemblyAfterFull.SumNanoseconds) + require.Equal(t, uint64(10*time.Millisecond), r.GroupPreparationAfterFull.SumNanoseconds) + require.Equal(t, uint64(20*time.Millisecond), r.TxLoopAfterFull.SumNanoseconds) + require.Equal(t, uint64(30*time.Millisecond), r.GroupJoinAssemblyAfterFull.SumNanoseconds+r.GroupPreparationAfterFull.SumNanoseconds+r.TxLoopAfterFull.SumNanoseconds) +} + +// One notification must pick up every ready contiguous batch, even when the +// remaining notifications are queued or the generation has just completed. +func TestStreamingEventRefreshesReadyBatches(t *testing.T) { + for _, completed := range []bool{false, true} { + t.Run(map[bool]string{false: "active", true: "completed"}[completed], func(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 9, 1700) + a, b, c := h.batch(t, 1, 3, txs[:3]), h.batch(t, 4, 6, txs[3:6]), h.batch(t, 7, 9, txs[6:]) + h.feed.pending[h.gen] = []*turbine.StreamBatch{c, a, b} + if completed { + h.feed.status[h.gen] = turbine.StreamDone + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamCompleted, Slot: a.Slot, Generation: h.gen}) + } else { + h.exec.handleEvent(h.event(a)) + } + sameTransactions(t, txs, h.executed()) + require.Len(t, h.exec.current.groups, 1) + h.exec.handleEvent(h.event(b)) + h.exec.handleEvent(h.event(c)) + require.Len(t, h.exec.current.groups, 1, "queued stale notifications do not execute twice") + }) + } +} + +func TestStreamingVerificationDeadlineAndDiscard(t *testing.T) { + old := StreamingExecutionCfg + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = old }() + h := newFinalizeFailureHarness(t) + cur := h.exec.current + h.exec.observed[cur.slot] = &streamingObservation{generation: h.gen} + StreamingExecutionCfg.MinGroupBatches = 2 + var stage string + cur.exec.setReplayStage = func(value string) { stage = value } + var firstContext context.Context + calls := 0 + h.exec.waitVerificationFn = func(ctx context.Context, batch *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) { + require.Equal(t, "streaming_sigverify_wait", stage) + deadline, ok := ctx.Deadline() + require.True(t, ok) + require.LessOrEqual(t, time.Until(deadline), streamingVerificationWait) + calls++ + if calls == 1 { + firstContext = ctx + return batch.WaitVerification(ctx) + } + require.Equal(t, firstContext, ctx, "one deadline covers all batches") + return nil, false, context.DeadlineExceeded + } + txs := transferTransactions(t, 2, 98123) + h.exec.handleEvent(h.event(h.batch(t, 4, 4, txs[:1]))) + require.Zero(t, calls) + h.exec.handleEvent(h.event(h.batch(t, 5, 5, txs[1:]))) + require.Equal(t, 2, calls) + h.assertUndone(t, "sigverify_timeout") + require.Equal(t, "streaming_wait", stage) + sameTransactions(t, h.txs[:3], cur.origin) + sameCopies(t, h.txs[:3], cur.exec.transactions) + _, ok, err := h.exec.finalize(h.env.exec.block, h.env.exec.parentBankSysvars) + require.NoError(t, err) + require.False(t, ok, "whole-block replay must take over") + metrics.GlobalBlockReplay.StreamingExecution = metrics.StreamingExecution{} + h.exec.noteWholeBlock(h.env.exec.block) + record := metrics.GlobalBlockReplay.StreamingExecution + require.Equal(t, "discarded:sigverify_timeout", record.NotOpenedReason) + require.Equal(t, uint64(2), record.VerificationWait.Count) + require.Positive(t, record.VerificationWait.SumNanoseconds) +} + +func TestStreamingVerificationWaitRespectsRemainingOpenAge(t *testing.T) { + old := StreamingExecutionCfg + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, MaxOpenAge: time.Second} + defer func() { StreamingExecutionCfg = old }() + h := newStreamingTestHarness(t) + cur := h.exec.current + cur.openedAt = time.Now().Add(-time.Second) + h.exec.waitVerificationFn = func(ctx context.Context, _ *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) { + deadline, ok := ctx.Deadline() + require.True(t, ok) + require.Equal(t, cur.openedAt.Add(time.Second), deadline) + require.ErrorIs(t, ctx.Err(), context.DeadlineExceeded) + return nil, false, ctx.Err() + } + h.exec.handleEvent(h.event(h.batch(t, 1, 1, transferTransactions(t, 1, 98124)))) + require.Nil(t, h.exec.current) + require.Empty(t, h.executed()) + require.Equal(t, "sigverify_timeout", h.discardReason()) +} + +func TestStreamingHeaderAdmissionBounds(t *testing.T) { + frontier := uint64(100) + s := newStreamingExecutor(streamingDeps{frontier: func() uint64 { return frontier }}) + for _, slot := range []uint64{100, 101, 132, 133, ^uint64(0)} { + g := turbine.NewDetachedStreamGeneration(slot) + s.rememberHeader(turbine.NewDetachedStreamMarker(g, 0, 0, turbine.StreamMarkerHeader, 100, solana.Hash{})) + } + require.Len(t, s.headers, 2) + require.Len(t, s.observed, 2) + require.Contains(t, s.headers, uint64(101)) + require.Contains(t, s.headers, uint64(132)) + frontier = ^uint64(0) - 1 + s.pruneHeaders(frontier) + g := turbine.NewDetachedStreamGeneration(^uint64(0)) + s.rememberHeader(turbine.NewDetachedStreamMarker(g, 0, 0, turbine.StreamMarkerHeader, frontier, solana.Hash{})) + require.Len(t, s.headers, 1, "distance comparison must not overflow") +} + +func TestStreamingDiscardPreservesLaterLeaderStakeEntries(t *testing.T) { + h := newFinalizeFailureHarness(t) + leaderSlot := h.exec.current.slot + 4 + leaderStake := solana.PublicKey{0xF7, 0xD1} + global.EnqueuePendingStakePubkey(leaderSlot, leaderStake) + t.Cleanup(func() { global.DropPendingStakePubkeys(leaderSlot) }) + h.exec.discard("test") + entries := global.PendingStakeEntriesSnapshot() + found := false + for _, entry := range entries { + if entry.Pubkey == leaderStake { + found = true + } + } + require.True(t, found, "discard must not remove the independent leader bank's stake entries") +} diff --git a/pkg/replay/transaction.go b/pkg/replay/transaction.go index ab5a61534..39af9c6e7 100644 --- a/pkg/replay/transaction.go +++ b/pkg/replay/transaction.go @@ -297,21 +297,106 @@ func recordVoteTimestampAndSlot(slotCtx *sealevel.SlotCtx, acct *accounts.Accoun func recordStakeAndVoteAccount(slotCtx *sealevel.SlotCtx, execCtx *sealevel.ExecutionCtx, acct *accounts.Account, modifiedVoteAccts bool) { if acct.Lamports == 0 || acct.Owner != a.VoteProgramAddr { - if global.VoteCacheItem(acct.Key) != nil { - global.DeleteVoteCacheItem(acct.Key) - markVoteStakeDirty(slotCtx.Slot) // global cache mutated — gates in-loop unwind + if voteCacheHas(slotCtx, acct.Key) { + deleteVoteCacheItem(slotCtx, acct.Key) + markSlotVoteStakeDirty(slotCtx) // global cache mutated — gates in-loop unwind } } else if modifiedVoteAccts { recordVoteTimestampAndSlot(slotCtx, acct) newVersionedVoteState, wasModified := execCtx.ModifiedVoteStates[acct.Key] if wasModified { - global.PutVoteCacheItem(acct.Key, newVersionedVoteState) + putVoteCacheItem(slotCtx, acct.Key, newVersionedVoteState) } - markVoteStakeDirty(slotCtx.Slot) + markSlotVoteStakeDirty(slotCtx) } if acct.Owner == a.StakeProgramAddr { recordStakeDelegation(slotCtx.Slot, acct) + markSlotVoteStakeDirty(slotCtx) + } +} + +// The vote cache is process-global. A bank executed speculatively (streaming +// replay) defers its puts/deletes into the SlotCtx so a discard leaves the +// cache untouched; publishDeferredVoteCache applies them once the bank is +// accepted. Non-deferring banks publish immediately, exactly as before. + +func voteCacheHas(slotCtx *sealevel.SlotCtx, key solana.PublicKey) bool { + if slotCtx.DeferVoteCachePublication { + slotCtx.PendingVoteCacheMu.Lock() + defer slotCtx.PendingVoteCacheMu.Unlock() + if _, deleted := slotCtx.PendingVoteCacheDeletes[key]; deleted { + return false + } + if _, pending := slotCtx.PendingVoteCache[key]; pending { + return true + } + } + return global.VoteCacheItem(key) != nil +} + +func putVoteCacheItem(slotCtx *sealevel.SlotCtx, key solana.PublicKey, state *sealevel.VoteStateVersions) { + if !slotCtx.DeferVoteCachePublication { + global.PutVoteCacheItem(key, state) + return + } + slotCtx.PendingVoteCacheMu.Lock() + defer slotCtx.PendingVoteCacheMu.Unlock() + if slotCtx.PendingVoteCache == nil { + slotCtx.PendingVoteCache = make(map[solana.PublicKey]*sealevel.VoteStateVersions) + } + slotCtx.PendingVoteCache[key] = state + delete(slotCtx.PendingVoteCacheDeletes, key) +} + +func deleteVoteCacheItem(slotCtx *sealevel.SlotCtx, key solana.PublicKey) { + if !slotCtx.DeferVoteCachePublication { + global.DeleteVoteCacheItem(key) + return + } + slotCtx.PendingVoteCacheMu.Lock() + defer slotCtx.PendingVoteCacheMu.Unlock() + if slotCtx.PendingVoteCacheDeletes == nil { + slotCtx.PendingVoteCacheDeletes = make(map[solana.PublicKey]struct{}) + } + slotCtx.PendingVoteCacheDeletes[key] = struct{}{} + delete(slotCtx.PendingVoteCache, key) +} + +func markSlotVoteStakeDirty(slotCtx *sealevel.SlotCtx) { + if slotCtx.DeferVoteCachePublication { + slotCtx.PendingVoteCacheMu.Lock() + slotCtx.VoteStakeDirty = true + slotCtx.PendingVoteCacheMu.Unlock() + return + } + markVoteStakeDirty(slotCtx.Slot) +} + +// publishDeferredVoteCache applies a speculative bank's buffered vote-cache +// changes and dirty marker. It is called by the streaming finalize step after +// the complete block has been matched to the executed prefix and is a no-op +// for banks that published immediately. +func publishDeferredVoteCache(slotCtx *sealevel.SlotCtx) { + if slotCtx == nil || !slotCtx.DeferVoteCachePublication { + return + } + slotCtx.PendingVoteCacheMu.Lock() + puts := slotCtx.PendingVoteCache + deletes := slotCtx.PendingVoteCacheDeletes + dirty := slotCtx.VoteStakeDirty + slotCtx.PendingVoteCache = nil + slotCtx.PendingVoteCacheDeletes = nil + slotCtx.VoteStakeDirty = false + slotCtx.DeferVoteCachePublication = false + slotCtx.PendingVoteCacheMu.Unlock() + for key := range deletes { + global.DeleteVoteCacheItem(key) + } + for key, state := range puts { + global.PutVoteCacheItem(key, state) + } + if dirty { markVoteStakeDirty(slotCtx.Slot) } } diff --git a/pkg/replay/transaction_preparation.go b/pkg/replay/transaction_preparation.go new file mode 100644 index 000000000..303dcc3e3 --- /dev/null +++ b/pkg/replay/transaction_preparation.go @@ -0,0 +1,133 @@ +package replay + +import ( + "crypto/sha256" + "encoding/binary" + "sort" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/fees" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// TransactionPreparer binds static transaction preparation to an immutable bank +// feature snapshot. Its source must remain immutable for the bank's lifetime. +// Prepared messages must also remain immutable, including their backing bytes. +// Account contents, transaction age and duplicate status are never cached here. +type TransactionPreparer struct { + source *features.Features + feats *features.Features + key [32]byte +} + +// PreparedTransaction contains only message/feature-derived data. Private fields +// prevent callers from substituting an unchecked hash, cost or instruction list. +type PreparedTransaction struct { + tx *solana.Transaction + key [32]byte + hash [32]byte + cost costmodel.TransactionCost + instrs []sealevel.Instruction + instructionAccts [][]sealevel.InstructionAccount + accountMetas []*solana.AccountMeta + limits *sealevel.ComputeBudgetLimits +} + +func NewTransactionPreparer(f *features.Features) *TransactionPreparer { + if f == nil { + return nil + } + clone := f.Clone() + gates := make([]features.FeatureGate, 0, len(*clone)) + for gate := range *clone { + gates = append(gates, gate) + } + sort.Slice(gates, func(i, j int) bool { + if gates[i].Address != gates[j].Address { + return string(gates[i].Address[:]) < string(gates[j].Address[:]) + } + return gates[i].Name < gates[j].Name + }) + h := sha256.New() + var value [8]byte + for _, gate := range gates { + info := (*clone)[gate] + h.Write(gate.Address[:]) + binary.LittleEndian.PutUint64(value[:], uint64(len(gate.Name))) + h.Write(value[:]) + h.Write([]byte(gate.Name)) + if info.Enabled { + h.Write([]byte{1}) + } else { + h.Write([]byte{0}) + } + binary.LittleEndian.PutUint64(value[:], info.ActivationSlot) + h.Write(value[:]) + } + p := &TransactionPreparer{source: f, feats: clone} + copy(p.key[:], h.Sum(nil)) + return p +} + +// Prepare returns nil on a static validation failure. Callers retain their normal +// processing path in that case, preserving its error classification and ordering. +func (p *TransactionPreparer) Prepare(tx *solana.Transaction) *PreparedTransaction { + if p == nil || tx == nil { + return nil + } + if tx.Message.GetVersion() == solana.MessageVersionV1 && !p.feats.IsActive(features.EnableTxV1) { + return nil + } + if txverify.SanitizeTransaction(tx) != nil { + return nil + } + if p.feats.IsActive(features.StaticInstructionLimit) && len(tx.Message.Instructions) > maxInstrTraceCapacity { + return nil + } + instrs, instructionAccts, metas, err := instrsAndAcctMetasFromTx(tx, p.feats) + if err != nil { + return nil + } + limits, err := sealevel.ComputeBudgetLimitsForTransaction(tx, instrs, p.feats) + if err != nil { + return nil + } + hash, err := TransactionMessageHash(tx) + if err != nil { + return nil + } + return &PreparedTransaction{tx: tx, key: p.key, hash: hash, + cost: costmodel.EstimatePreparedTransactionCost(tx, instrs, limits, p.feats), + instrs: instrs, instructionAccts: instructionAccts, accountMetas: metas, limits: limits} +} + +func (p *TransactionPreparer) Matches(prepared *PreparedTransaction, tx *solana.Transaction, f *features.Features) bool { + return p != nil && p.source == f && prepared != nil && prepared.tx == tx && p.key == prepared.key +} + +func (p *PreparedTransaction) MessageHash() [32]byte { return p.hash } + +// Cost returns the estimate; its writable-account slice is read-only. +func (p *PreparedTransaction) Cost() costmodel.TransactionCost { return p.cost } + +// LoadAndExecute reuses preparation only for the same immutable message and a +// matching feature snapshot. All bank-dependent checks still run on every call. +func (p *TransactionPreparer) LoadAndExecute(input LoadAndExecuteTransactionInput, prepared *PreparedTransaction) LoadAndExecuteTransactionOutput { + if input.SlotCtx == nil || p == nil || p.source != input.SlotCtx.Features || !p.Matches(prepared, input.Transaction, input.SlotCtx.Features) { + return LoadAndExecuteTransaction(input) + } + return loadAndExecuteTransaction(input, prepared) +} + +// PayerCanFund keeps strict leader admission, using current payer state while +// sharing the already-validated instructions and compute limits. +func (p *TransactionPreparer) PayerCanFund(slotCtx *sealevel.SlotCtx, tx *solana.Transaction, prepared *PreparedTransaction) error { + if slotCtx == nil || !p.Matches(prepared, tx, slotCtx.Features) { + return fees.PayerCanFund(slotCtx, tx) + } + _, err := fees.ValidateTransactionFeePayer(slotCtx, tx, prepared.instrs, prepared.limits) + return err +} diff --git a/pkg/replay/transaction_processing_pure.go b/pkg/replay/transaction_processing_pure.go index 061703db1..434346460 100644 --- a/pkg/replay/transaction_processing_pure.go +++ b/pkg/replay/transaction_processing_pure.go @@ -3,7 +3,6 @@ package replay import ( "errors" "math" - "time" "github.com/Overclock-Validator/mithril/pkg/accounts" "github.com/Overclock-Validator/mithril/pkg/arena" @@ -39,6 +38,10 @@ type LoadAndExecuteTransactionInput struct { // banks also skip writable-account result materialization; RPC simulation // leaves this false to retain the rich result. LeanResult bool + // SkipTimingMetrics omits the detailed transaction and instruction-dispatch + // timings for leader execution. Replay and simulation retain their default + // instrumentation; program-specific instrumentation is independent. + SkipTimingMetrics bool // CapturePreBalances retains pre-fee balances in lean mode. Rich mode // always captures them for RPC compatibility. CapturePreBalances bool @@ -124,6 +127,10 @@ func feeOnlyRollbackAccountsDataSize(slotCtx *sealevel.SlotCtx, tx *solana.Trans } func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExecuteTransactionOutput { + return loadAndExecuteTransaction(input, nil) +} + +func loadAndExecuteTransaction(input LoadAndExecuteTransactionInput, prepared *PreparedTransaction) LoadAndExecuteTransactionOutput { tx := input.Transaction slotCtx := input.SlotCtx @@ -131,64 +138,76 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec input.Arena.Reset() } - if tx == nil || slotCtx == nil || slotCtx.Features == nil { - return sanitizeFailureOutput() - } + var instrs []sealevel.Instruction + var instructionAcctsPerInstr [][]sealevel.InstructionAccount + var txAcctMetas []*solana.AccountMeta + var computeBudgetLimits *sealevel.ComputeBudgetLimits + var err error + start := metrics.StartTiming(false) + if prepared != nil { + instrs, instructionAcctsPerInstr, txAcctMetas = prepared.instrs, prepared.instructionAccts, prepared.accountMetas + computeBudgetLimits = prepared.limits + } else { + if tx == nil || slotCtx == nil || slotCtx.Features == nil { + return sanitizeFailureOutput() + } - // Match Agave's Bank verification order: once a transaction has decoded as - // v1, the feature gate is checked before structural sanitization. - if tx.Message.GetVersion() == solana.MessageVersionV1 && !slotCtx.Features.IsActive(features.EnableTxV1) { - return LoadAndExecuteTransactionOutput{ - ProcessingResult: TransactionProcessingResult{ - TransactionError: &TransactionError{ - ErrorType: TransactionErrorUnsupportedVersion, - InstructionError: TxErrUnsupportedVersion, + // Match Agave's Bank verification order: once a transaction has decoded as + // v1, the feature gate is checked before structural sanitization. + if tx.Message.GetVersion() == solana.MessageVersionV1 && !slotCtx.Features.IsActive(features.EnableTxV1) { + return LoadAndExecuteTransactionOutput{ + ProcessingResult: TransactionProcessingResult{ + TransactionError: &TransactionError{ + ErrorType: TransactionErrorUnsupportedVersion, + InstructionError: TxErrUnsupportedVersion, + }, }, - }, + } + } + // Reject malformed transactions before account-indexed code can observe + // them. The helper also handles Mithril's already-resolved v0 messages. + if err := txverify.SanitizeTransaction(tx); err != nil { + return sanitizeFailureOutput() + } + // Mirror block-replay's StaticInstructionLimit cap so pre-activation + // clusters fail mid-execution like Agave instead of SanitizeFailure. + if slotCtx.Features.IsActive(features.StaticInstructionLimit) && + len(tx.Message.Instructions) > maxInstrTraceCapacity { + return sanitizeFailureOutput() } - } - // Reject malformed transactions before account-indexed code can observe - // them. The helper also handles Mithril's already-resolved v0 messages. - if err := txverify.SanitizeTransaction(tx); err != nil { - return sanitizeFailureOutput() - } - // Mirror block-replay's StaticInstructionLimit cap so pre-activation - // clusters fail mid-execution like Agave instead of SanitizeFailure. - if slotCtx.Features.IsActive(features.StaticInstructionLimit) && - len(tx.Message.Instructions) > maxInstrTraceCapacity { - return sanitizeFailureOutput() - } - // Parse instructions and account metas - start := time.Now() - instrs, instructionAcctsPerInstr, txAcctMetas, err := instrsAndAcctMetasFromTx(tx, slotCtx.Features) - if err != nil { - return LoadAndExecuteTransactionOutput{ - ProcessingResult: TransactionProcessingResult{ - TransactionError: &TransactionError{ - ErrorType: TransactionErrorSanitizeFailure, - InstructionError: err, + // Parse instructions and account metas + start = metrics.StartTiming(!input.SkipTimingMetrics) + instrs, instructionAcctsPerInstr, txAcctMetas, err = instrsAndAcctMetasFromTx(tx, slotCtx.Features) + if err != nil { + return LoadAndExecuteTransactionOutput{ + ProcessingResult: TransactionProcessingResult{ + TransactionError: &TransactionError{ + ErrorType: TransactionErrorSanitizeFailure, + InstructionError: err, + }, }, - }, + } } - } - metrics.GlobalBlockReplay.InstructionsAndAccountMetasFromTx.AddTimingSince(start) - - // Compute budget limits - start = time.Now() - computeBudgetLimits, err := sealevel.ComputeBudgetLimitsForTransaction(tx, instrs, slotCtx.Features) - if err != nil { - return LoadAndExecuteTransactionOutput{ - ProcessingResult: TransactionProcessingResult{ - TransactionError: &TransactionError{ - ErrorType: TransactionErrorInstructionError, - InstructionError: err, + metrics.GlobalBlockReplay.InstructionsAndAccountMetasFromTx.AddTimingSince(start) + + // Compute budget limits + start = metrics.StartTiming(!input.SkipTimingMetrics) + computeBudgetLimits, err = sealevel.ComputeBudgetLimitsForTransaction(tx, instrs, slotCtx.Features) + if err != nil { + return LoadAndExecuteTransactionOutput{ + ProcessingResult: TransactionProcessingResult{ + TransactionError: &TransactionError{ + ErrorType: TransactionErrorInstructionError, + InstructionError: err, + }, }, - }, - Instrs: instrs, + Instrs: instrs, + } } + metrics.GlobalBlockReplay.ComputeBudgetExecutionInstructions.AddTimingSince(start) + } - metrics.GlobalBlockReplay.ComputeBudgetExecutionInstructions.AddTimingSince(start) // Validate transaction age if !sealevel.IsTransactionAgeValid(tx, instrs, slotCtx) { @@ -236,7 +255,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec } // Load and validate accounts - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) instructionsSysvarIdx := instructionsSysvarAccountIndex(tx) var instrsAcct *accounts.Account if instructionsSysvarIdx >= 0 { @@ -303,6 +322,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec execCtx.TransactionContext.Signature = tx.Signatures[0] execCtx.TransactionContext.BorrowedAccountArena = input.Arena execCtx.IsSimulation = input.IsSimulation + execCtx.SkipTimingMetrics = input.SkipTimingMetrics execCtx.RecordInnerInstructions = input.RecordInnerInstructions // Capture pre-balance lamports (before fee deduction) @@ -332,7 +352,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec // Calculate and deduct fees. RentForSlot supplies the exemption // minimum so a rent-exempt payer is rejected before instructions run. - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) txFeeInfo, _, err := fees.CalculateAndDeductTxFees(tx, input.TxMeta, instrs, &execCtx.TransactionContext.Accounts, computeBudgetLimits, slotCtx.Features, fees.RentForSlot(slotCtx), input.IsSimulation) if err != nil { errType, accountIndex := feePayerTransactionError(err) @@ -355,7 +375,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec metrics.GlobalBlockReplay.CalcAndDeductFees.AddTimingSince(start) // Read rent sysvar - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) rentSysvar, err := sealevel.ReadRentSysvar(execCtx) if err != nil { // Rent sysvar unreadable; return cleanly so the RPC worker @@ -374,18 +394,18 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec metrics.GlobalBlockReplay.ReadRentSysvar.AddTimingSince(start) // Set rent-exempt rent epoch max and compute pre-tx rent states - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) rent.MaybeSetRentExemptRentEpochMax(slotCtx, &rentSysvar, &execCtx.Features, &execCtx.TransactionContext.Accounts) preTxRentStates := rent.NewRentStateInfo(&rentSysvar, execCtx.TransactionContext, &execCtx.Features) metrics.GlobalBlockReplay.PreTxRentStates.AddTimingSince(start) // Execute all instructions var instrErr error - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) for instrIdx, instr := range tx.Message.Instructions { execCtx.SetCurrentTopLevelInstr(uint8(instrIdx)) if instructionsSysvarIdx >= 0 { - ixStart := time.Now() + ixStart := metrics.StartTiming(!input.SkipTimingMetrics) err = fixupInstructionsSysvarAcct(execCtx, instructionsSysvarIdx, uint16(instrIdx)) if err != nil { instrErr = err @@ -421,7 +441,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec metrics.GlobalBlockReplay.IxLoop.AddTimingSince(start) // Check rent state transitions - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) postTxRentStates := rent.NewRentStateInfo(&rentSysvar, execCtx.TransactionContext, &execCtx.Features) rentStateErr := rent.VerifyRentStateChanges(preTxRentStates, postTxRentStates, execCtx.TransactionContext) metrics.GlobalBlockReplay.PostTxRentStates.AddTimingSince(start) diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index 8320df056..85068b084 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -10,6 +10,7 @@ import ( "path/filepath" "sort" "sync" + "sync/atomic" b "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/state" @@ -48,6 +49,38 @@ type transactionStatusNode struct { hasBlockID bool parent *transactionStatusNode delta transactionStatusDelta + + // Shared by parentless checkpoint copies and relinked retained nodes. + // Only the memoized encoding changes after publication; lineage and delta + // remain immutable. Never copy the atomic field after its first use. + encoding atomic.Pointer[transactionStatusNodeEncoding] +} + +type transactionStatusNodeEncoding struct { + once sync.Once + data []byte +} + +func (n *transactionStatusNode) encodingCache() *transactionStatusNodeEncoding { + if cache := n.encoding.Load(); cache != nil { + return cache + } + cache := new(transactionStatusNodeEncoding) + if n.encoding.CompareAndSwap(nil, cache) { + return cache + } + return n.encoding.Load() +} + +// copyInto initializes a fresh node, sharing its encoding without retaining +// excluded ancestry or copying a used synchronization primitive. The encoding excludes +// parent links and depends only on the immutable slot, block ID and delta. +func (n *transactionStatusNode) copyInto(copy *transactionStatusNode, parent *transactionStatusNode) { + *copy = transactionStatusNode{ + slot: n.slot, blockID: n.blockID, hasBlockID: n.hasBlockID, + parent: parent, delta: n.delta, + } + copy.encoding.Store(n.encodingCache()) } type visibleTransactionStatusGroup struct { @@ -72,6 +105,9 @@ type TransactionStatusCache struct { // from a known-empty genesis cache. Without this bit, completeness requires // the full 300 retained roots; a serialized boolean alone is not evidence. coverageFromGenesis bool + + // Protected by mu; see transactionStatusValidation. Never serialized. + validationVersion uint64 } // TransactionStatusView is an immutable view of one bank lineage. It lazily @@ -412,6 +448,7 @@ func (c *TransactionStatusCache) BindTipBlockID(slot uint64, blockID solana.Hash if c.tip.hasBlockID && c.tip.blockID != blockID { return fmt.Errorf("transaction status tip at slot %d has block id %s, cannot bind %s", slot, c.tip.blockID, blockID) } + c.invalidateValidationLocked() c.tip = &transactionStatusNode{ slot: slot, blockID: blockID, hasBlockID: true, parent: c.tip.parent, delta: c.tip.delta, @@ -515,25 +552,52 @@ func (c *TransactionStatusCache) ValidateBlock(block *b.Block) error { // validateBlockWithPlan preserves the status-cache checks while letting // replay reuse the exact immutable identities used for execution planning. func (c *TransactionStatusCache) validateBlockWithPlan(block *b.Block, plan blockTransactionExecutionPlan) error { + _, err := c.validateBlockForPublication(block, plan) + return err +} + +func (c *TransactionStatusCache) validateBlockForPublication(block *b.Block, plan blockTransactionExecutionPlan) (transactionStatusValidation, error) { if block == nil { - return errors.New("nil block") + return transactionStatusValidation{}, errors.New("nil block") } if plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { - return errors.New("prepared transaction message identities do not match block") + return transactionStatusValidation{}, errors.New("prepared transaction message identities do not match block") } if c == nil { - return &IncompleteTransactionStatusCoverageError{} + return transactionStatusValidation{}, &IncompleteTransactionStatusCoverageError{} } c.mu.RLock() defer c.mu.RUnlock() if !c.coverageComplete { - return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} + return transactionStatusValidation{}, &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} } if err := c.validateParentLocked(block); err != nil { - return err + return transactionStatusValidation{}, err } - return c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities) + if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { + return transactionStatusValidation{}, err + } + return transactionStatusValidation{cache: c, identities: plan.messageIdentities, version: c.validationVersion}, nil +} + +// validateTransactionsAgainstAncestors is the per-group form of the ancestor +// already-processed check, for execution that starts before the complete +// block exists. It does not validate the parent link; the complete block is +// validated again in full, with validateBlockForPublication, before commit. +func (c *TransactionStatusCache) validateTransactionsAgainstAncestors(slot uint64, identities *b.PreparedTransactionMessageIdentities) error { + if identities == nil { + return errors.New("nil transaction message identities") + } + if c == nil { + return &IncompleteTransactionStatusCoverageError{} + } + c.mu.RLock() + defer c.mu.RUnlock() + if !c.coverageComplete { + return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} + } + return c.validateAncestorTransactionsLocked(slot, identities) } func (c *TransactionStatusCache) validateAncestorTransactionsLocked(slot uint64, identities *b.PreparedTransactionMessageIdentities) error { @@ -586,6 +650,14 @@ func (c *TransactionStatusCache) CommitBlock(block *b.Block) error { // commitBlockWithPlan atomically rechecks the mutable lineage/status state and // publishes the already-prepared immutable transaction identities. func (c *TransactionStatusCache) commitBlockWithPlan(block *b.Block, plan blockTransactionExecutionPlan) error { + return c.commitBlockWithPreparedDelta(block, plan, nil) +} + +func (c *TransactionStatusCache) commitBlockWithPreparedDelta(block *b.Block, plan blockTransactionExecutionPlan, prepared *preparedTransactionStatusDelta) error { + return c.commitBlockWithValidation(block, plan, prepared, transactionStatusValidation{}) +} + +func (c *TransactionStatusCache) commitBlockWithValidation(block *b.Block, plan blockTransactionExecutionPlan, prepared *preparedTransactionStatusDelta, validation transactionStatusValidation) error { if block == nil || plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { return errors.New("prepared transaction message identities do not match block") } @@ -594,33 +666,45 @@ func (c *TransactionStatusCache) commitBlockWithPlan(block *b.Block, plan blockT if !c.coverageComplete { return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} } - // Parent lineage and ancestor status are mutable, so both remain under the - // publication lock even when hashing and same-bank deduplication happened - // earlier. This keeps commit safe across a concurrent branch transition. + // Always check coverage, block binding and parent lineage. Reuse the earlier + // ancestor scan only under this lock and only for the same unchanged cache + // and immutable identities. A branch transition (including away and back) + // or root/prune invalidates it, requiring a fresh scan before publication. if err := c.validateParentLocked(block); err != nil { return err } - if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { - return err + if !validation.reusableForLocked(c, plan.messageIdentities) { + if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { + return err + } } - delta := make(transactionStatusDelta) - for index := 0; index < plan.messageIdentities.Len(); index++ { - identity := plan.messageIdentities.Identity(index) - blockhash := identity.RecentBlockhash - group := delta[blockhash] - if group == nil { - keyIndex := uint8(0) + delta := transactionStatusDelta(nil) + if prepared != nil && prepared.identities == plan.messageIdentities { + delta = prepared.delta + // A restore or branch transition can change a blockhash's slice offset. + // Rebuild from full identities if a group changed offset or disappeared; + // a missing group uses the same zero offset as fresh preparation. + for blockhash, group := range delta { + index := uint8(0) if visible := c.visible[blockhash]; visible != nil { - keyIndex = visible.keyIndex + index = visible.keyIndex } - group = &transactionStatusGroup{ - keyIndex: keyIndex, - keys: make(map[transactionStatusKey]struct{}), + if index != group.keyIndex { + delta = nil + break } - delta[blockhash] = group } - group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + } + if delta == nil { + counts := countTransactionStatusGroups(plan.messageIdentities) + indexes := make(map[solana.Hash]uint8, len(counts)) + for blockhash := range counts { + if visible := c.visible[blockhash]; visible != nil { + indexes[blockhash] = visible.keyIndex + } + } + delta = buildTransactionStatusDelta(plan.messageIdentities, counts, indexes) } if err := c.addDeltaVisibleLocked(delta); err != nil { @@ -663,6 +747,7 @@ func (c *TransactionStatusCache) Root(through uint64) bool { } c.mu.Lock() defer c.mu.Unlock() + c.invalidateValidationLocked() wasComplete := c.coverageComplete newlyRooted := c.countNodesBetweenLocked(c.rootedThrough, through) if through > c.rootedThrough { @@ -681,10 +766,32 @@ func (c *TransactionStatusCache) Root(through uint64) bool { return !wasComplete && c.coverageComplete } -// SnapshotThrough serializes only the rooted lineage needed at through. It is -// called while constructing a fold job, so the blob rides in that exact durable -// manifest without being copied into every speculative ResumeContext. -func (c *TransactionStatusCache) SnapshotThrough(through uint64) ([]byte, error) { +// TransactionStatusSnapshot pins an immutable checkpoint view. MarshalBinary +// must use only captured data, without locking or revisiting the live cache, +// and return an owned payload. It can run on the checkpoint worker while replay +// commits, roots, or unwinds its current lineage. +type TransactionStatusSnapshot interface { + MarshalBinary() ([]byte, error) +} + +type transactionStatusSnapshot struct { + nodes []*transactionStatusNode + rootedSinceSeed uint16 + complete bool + coverageFromGenesis bool +} + +func (s *transactionStatusSnapshot) MarshalBinary() ([]byte, error) { + if s == nil { + return nil, nil + } + return marshalTransactionStatusNodes(s.nodes, s.rootedSinceSeed, s.complete, s.coverageFromGenesis) +} + +// CaptureSnapshotThrough selects the exact checkpoint lineage and coverage on +// replay, but leaves transaction-key sorting and serialization to the worker. +// Published deltas are immutable; only small node headers are copied here. +func (c *TransactionStatusCache) CaptureSnapshotThrough(through uint64) (TransactionStatusSnapshot, error) { if c == nil { return nil, nil } @@ -700,7 +807,27 @@ func (c *TransactionStatusCache) SnapshotThrough(through uint64) ([]byte, error) if rootedSinceSeed > maxTransactionStatusRoots { rootedSinceSeed = maxTransactionStatusRoots } - return marshalTransactionStatusNodes(nodes, uint16(rootedSinceSeed), complete, c.coverageFromGenesis) + owned := make([]transactionStatusNode, len(nodes)) + pinned := make([]*transactionStatusNode, len(nodes)) + for i, node := range nodes { + // Do not keep the old parent chain or copy its atomic field. + node.copyInto(&owned[i], nil) + pinned[i] = &owned[i] + } + return &transactionStatusSnapshot{ + nodes: pinned, rootedSinceSeed: uint16(rootedSinceSeed), complete: complete, + coverageFromGenesis: c.coverageFromGenesis, + }, nil +} + +// SnapshotThrough is the synchronous convenience API. Serialization still +// happens after releasing the cache lock; normal folds use CaptureSnapshotThrough. +func (c *TransactionStatusCache) SnapshotThrough(through uint64) ([]byte, error) { + snapshot, err := c.CaptureSnapshotThrough(through) + if err != nil || snapshot == nil { + return nil, err + } + return snapshot.MarshalBinary() } func (c *TransactionStatusCache) processedSlotLocked(blockhash solana.Hash, key transactionStatusKey) uint64 { @@ -749,6 +876,7 @@ func (c *TransactionStatusCache) validateParentLocked(block *b.Block) error { } func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDelta) error { + c.invalidateValidationLocked() for blockhash, deltaGroup := range delta { if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { return fmt.Errorf("transaction status blockhash %s uses inconsistent key indexes %d and %d", @@ -760,7 +888,7 @@ func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDe if group == nil { group = &visibleTransactionStatusGroup{ keyIndex: deltaGroup.keyIndex, - keys: make(map[transactionStatusKey]uint16), + keys: make(map[transactionStatusKey]uint16, len(deltaGroup.keys)), } c.visible[blockhash] = group } @@ -772,6 +900,7 @@ func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDe } func (c *TransactionStatusCache) removeDeltaVisibleLocked(delta transactionStatusDelta) { + c.invalidateValidationLocked() for blockhash, deltaGroup := range delta { group := c.visible[blockhash] if group == nil { @@ -839,20 +968,76 @@ func (c *TransactionStatusCache) pruneLocked(through uint64) { if drop <= 0 { return } - for _, node := range nodes[:drop] { - c.removeDeltaVisibleLocked(node.delta) - } retained := nodes[drop:] + c.expireVisibleLocked(nodes[:drop], retained) var parent *transactionStatusNode for _, old := range retained { - parent = &transactionStatusNode{ - slot: old.slot, blockID: old.blockID, hasBlockID: old.hasBlockID, - parent: parent, delta: old.delta, - } + next := new(transactionStatusNode) + old.copyInto(next, parent) + parent = next } c.tip = parent } +// expireVisibleLocked expires a whole rooted batch. Most old blockhash groups +// have no surviving bank and can be removed without visiting their transaction +// keys. For a group crossing the boundary, update whichever side is smaller. +// Immutable node deltas (including those pinned by producer views/checkpoints) +// are never mutated. Unrooted retained banks count as survivors too. +func (c *TransactionStatusCache) expireVisibleLocked(expired, retained []*transactionStatusNode) { + type groupExpiry struct { + expiredKeys int + retainedKeys int + survivors []*transactionStatusGroup + } + groups := make(map[solana.Hash]*groupExpiry) + for _, node := range expired { + for hash, delta := range node.delta { + g := groups[hash] + if g == nil { + g = &groupExpiry{} + groups[hash] = g + } + g.expiredKeys += len(delta.keys) + } + } + for _, node := range retained { + for hash, delta := range node.delta { + if g := groups[hash]; g != nil { + g.retainedKeys += len(delta.keys) + g.survivors = append(g.survivors, delta) + } + } + } + for hash, g := range groups { + if len(g.survivors) == 0 { + delete(c.visible, hash) + } else if g.retainedKeys < g.expiredKeys { + rebuilt := &visibleTransactionStatusGroup{keyIndex: g.survivors[0].keyIndex, keys: make(map[transactionStatusKey]uint16)} + for _, delta := range g.survivors { + for key := range delta.keys { + rebuilt.keys[key]++ + } + } + c.visible[hash] = rebuilt + if len(rebuilt.keys) == 0 { + delete(c.visible, hash) + } + } + } + for _, node := range expired { + for hash, delta := range node.delta { + g := groups[hash] + if len(g.survivors) == 0 || g.retainedKeys < g.expiredKeys { + continue + } + // The existing removal path preserves reference counts for keys + // occurring in more than one retained/expired bank. + c.removeDeltaVisibleLocked(transactionStatusDelta{hash: delta}) + } + } +} + func sliceTransactionStatusKey(messageHash [32]byte, keyIndex uint8) transactionStatusKey { // Match Agave's saturating_sub(CACHED_KEY_SIZE + 1), including its // deliberate exclusion of the final possible starting offset. @@ -867,7 +1052,16 @@ func sliceTransactionStatusKey(messageHash [32]byte, keyIndex uint8) transaction } func marshalTransactionStatusNodes(nodes []*transactionStatusNode, rootedSinceSeed uint16, complete bool, coverageFromGenesis bool) ([]byte, error) { - var buf bytes.Buffer + encoded := make([][]byte, len(nodes)) + size := 9 // magic, flags, rooted count and node count + for i, node := range nodes { + cache := node.encodingCache() + cache.once.Do(func() { cache.data = marshalTransactionStatusNode(node) }) + encoded[i] = cache.data + size += len(cache.data) + } + // Every caller owns its result. Never return or append into a cached slice. + buf := bytes.NewBuffer(make([]byte, 0, size)) buf.Write(transactionStatusSnapshotMagic[:]) flags := byte(0) if complete { @@ -877,47 +1071,61 @@ func marshalTransactionStatusNodes(nodes []*transactionStatusNode, rootedSinceSe flags |= 2 } buf.WriteByte(flags) - _ = binary.Write(&buf, binary.LittleEndian, rootedSinceSeed) - _ = binary.Write(&buf, binary.LittleEndian, uint16(len(nodes))) - for _, node := range nodes { - _ = binary.Write(&buf, binary.LittleEndian, node.slot) - nodeFlags := byte(0) - if node.hasBlockID { - nodeFlags = 1 - } - buf.WriteByte(nodeFlags) - if node.hasBlockID { - buf.Write(node.blockID[:]) - } - blockhashes := make([]solana.Hash, 0, len(node.delta)) - for blockhash := range node.delta { - blockhashes = append(blockhashes, blockhash) + _ = binary.Write(buf, binary.LittleEndian, rootedSinceSeed) + _ = binary.Write(buf, binary.LittleEndian, uint16(len(nodes))) + for _, data := range encoded { + buf.Write(data) + } + return buf.Bytes(), nil +} + +func marshalTransactionStatusNode(node *transactionStatusNode) []byte { + size := 8 + 1 + 4 // slot, flags and group count + if node.hasBlockID { + size += len(node.blockID) + } + for _, group := range node.delta { + size += 32 + 1 + 4 + transactionStatusKeySize*len(group.keys) + } + buf := bytes.NewBuffer(make([]byte, 0, size)) + _ = binary.Write(buf, binary.LittleEndian, node.slot) + nodeFlags := byte(0) + if node.hasBlockID { + nodeFlags = 1 + } + buf.WriteByte(nodeFlags) + if node.hasBlockID { + buf.Write(node.blockID[:]) + } + blockhashes := make([]solana.Hash, 0, len(node.delta)) + for blockhash := range node.delta { + blockhashes = append(blockhashes, blockhash) + } + sort.Slice(blockhashes, func(i, j int) bool { + return bytes.Compare(blockhashes[i][:], blockhashes[j][:]) < 0 + }) + _ = binary.Write(buf, binary.LittleEndian, uint32(len(blockhashes))) + for _, blockhash := range blockhashes { + group := node.delta[blockhash] + buf.Write(blockhash[:]) + buf.WriteByte(group.keyIndex) + keys := make([]transactionStatusKey, 0, len(group.keys)) + for key := range group.keys { + keys = append(keys, key) } - sort.Slice(blockhashes, func(i, j int) bool { - return bytes.Compare(blockhashes[i][:], blockhashes[j][:]) < 0 + sort.Slice(keys, func(i, j int) bool { + return bytes.Compare(keys[i][:], keys[j][:]) < 0 }) - _ = binary.Write(&buf, binary.LittleEndian, uint32(len(blockhashes))) - for _, blockhash := range blockhashes { - group := node.delta[blockhash] - buf.Write(blockhash[:]) - buf.WriteByte(group.keyIndex) - keys := make([]transactionStatusKey, 0, len(group.keys)) - for key := range group.keys { - keys = append(keys, key) - } - sort.Slice(keys, func(i, j int) bool { - return bytes.Compare(keys[i][:], keys[j][:]) < 0 - }) - _ = binary.Write(&buf, binary.LittleEndian, uint32(len(keys))) - for _, key := range keys { - buf.Write(key[:]) - } + _ = binary.Write(buf, binary.LittleEndian, uint32(len(keys))) + for _, key := range keys { + buf.Write(key[:]) } } - return buf.Bytes(), nil + return buf.Bytes() } func (c *TransactionStatusCache) restore(data []byte) error { + c.invalidateValidationLocked() reader := bytes.NewReader(data) var magic [4]byte if _, err := io.ReadFull(reader, magic[:]); err != nil { diff --git a/pkg/replay/transaction_status_capture_bench_test.go b/pkg/replay/transaction_status_capture_bench_test.go new file mode 100644 index 000000000..234743222 --- /dev/null +++ b/pkg/replay/transaction_status_capture_bench_test.go @@ -0,0 +1,113 @@ +package replay + +import ( + "crypto/sha256" + "encoding/binary" + "fmt" + "testing" + + "github.com/gagliardetto/solana-go" +) + +var checkpointBenchmarkPayload []byte +var checkpointBenchmarkCapture TransactionStatusSnapshot + +func checkpointEncodingFixture() *TransactionStatusCache { + // A private, not-yet-published fixture with the same complete 300-root + // metadata as an imported cache. 1.5 million keys encode to roughly 30 MB. + c := newTransactionStatusCache(true) + c.coverageFromGenesis = false + c.rootedSinceSeed = maxTransactionStatusRoots + c.rootedThrough = maxTransactionStatusRoots + for slot := uint64(1); slot <= maxTransactionStatusRoots; slot++ { + keys := make(map[transactionStatusKey]struct{}, 5000) + var seed [16]byte + binary.LittleEndian.PutUint64(seed[:8], slot) + for i := uint64(0); i < 5000; i++ { + binary.LittleEndian.PutUint64(seed[8:], i) + hash := sha256.Sum256(seed[:]) + var key transactionStatusKey + copy(key[:], hash[:]) + keys[key] = struct{}{} + } + c.tip = &transactionStatusNode{slot: slot, parent: c.tip, + delta: transactionStatusDelta{solana.Hash{1}: {keyIndex: 7, keys: keys}}} + } + return c +} + +func BenchmarkTransactionStatusCheckpointCapture(b *testing.B) { + c := checkpointEncodingFixture() + view, err := c.CaptureSnapshotThrough(maxTransactionStatusRoots) + if err != nil { + b.Fatal(err) + } + b.Run("SynchronousBaseline", func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + checkpointBenchmarkPayload, err = legacyStatusSnapshotForTest(c, maxTransactionStatusRoots) + if err != nil { + b.Fatal(err) + } + } + }) + b.Run("CaptureOnReplay", func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + checkpointBenchmarkCapture, err = c.CaptureSnapshotThrough(maxTransactionStatusRoots) + if err != nil { + b.Fatal(err) + } + } + }) + b.Run("EncodeOnWorker", func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + checkpointBenchmarkPayload, err = view.MarshalBinary() + if err != nil { + b.Fatal(err) + } + } + }) +} + +// Moving 300-root windows at several checkpoint cadences. Fixtures and initial +// cache warming are excluded; allocating new node headers and encoding/output +// allocation are included. This measures encoding only, not fsync or account +// checkpoint work. Cold encodes represent first startup/all-new windows. +func BenchmarkTransactionStatusCheckpointEncoding(b *testing.B) { + c := checkpointEncodingFixture() + view, err := c.CaptureSnapshotThrough(300) + if err != nil { + b.Fatal(err) + } + seed := view.(*transactionStatusSnapshot).nodes + for _, advance := range []int{0, 1, 8, 32, defaultFoldBatchSlots, 300} { + for _, cached := range []bool{false, true} { + b.Run(fmt.Sprintf("new=%d/cached=%t", advance, cached), func(b *testing.B) { + nodes := append([]*transactionStatusNode(nil), seed...) + if cached { + _, _ = marshalTransactionStatusNodes(nodes, 300, true, false) + } + b.ReportAllocs() + b.ResetTimer() + for n := 0; n < b.N; n++ { + copy(nodes, nodes[advance:]) + for i := 300 - advance; i < 300; i++ { + nodes[i] = &transactionStatusNode{slot: uint64(301 + n*advance + i), delta: seed[i].delta} + } + var err error + if cached { + checkpointBenchmarkPayload, err = marshalTransactionStatusNodes(nodes, 300, true, false) + } else { + checkpointBenchmarkPayload, err = marshalTransactionStatusNodesUncached(nodes, 300, true, false) + } + if err != nil { + b.Fatal(err) + } + } + b.SetBytes(int64(len(checkpointBenchmarkPayload))) + }) + } + } +} diff --git a/pkg/replay/transaction_status_capture_test.go b/pkg/replay/transaction_status_capture_test.go new file mode 100644 index 000000000..34b8273af --- /dev/null +++ b/pkg/replay/transaction_status_capture_test.go @@ -0,0 +1,334 @@ +package replay + +import ( + "bytes" + "encoding/binary" + "fmt" + "math/rand" + "sort" + "sync" + "testing" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// Keep the pre-split selection/metadata calculation as a differential oracle. +// Use the original uncached wire encoder to check byte-for-byte compatibility. +func legacyStatusSnapshotForTest(c *TransactionStatusCache, through uint64) ([]byte, error) { + c.mu.RLock() + defer c.mu.RUnlock() + nodes := c.nodesThroughLocked(through) + if len(nodes) > maxTransactionStatusRoots { + nodes = nodes[len(nodes)-maxTransactionStatusRoots:] + } + rooted := uint32(c.rootedSinceSeed) + uint32(c.countNodesBetweenLocked(c.rootedThrough, through)) + complete := c.coverageComplete || rooted >= maxTransactionStatusRoots + if rooted > maxTransactionStatusRoots { + rooted = maxTransactionStatusRoots + } + return marshalTransactionStatusNodesUncached(nodes, uint16(rooted), complete, c.coverageFromGenesis) +} + +func importedStatusCacheForTest(t *testing.T) *TransactionStatusCache { + t.Helper() + roots := make([]txstatus.SnapshotSlotDelta, maxTransactionStatusRoots) + for i := range roots { + roots[i] = txstatus.SnapshotSlotDelta{Slot: uint64(i + 1), IsRoot: true} + } + c, err := NewTransactionStatusCacheFromAgaveSnapshot(roots, maxTransactionStatusRoots) + require.NoError(t, err) + return c +} + +func captureTestBlock(slot uint64, branch byte) *b.Block { + tx := statusCacheTestTransaction(1, 2, branch) + data := make([]byte, 9) + binary.LittleEndian.PutUint64(data, slot) + data[8] = branch + tx.Message.Instructions[0].Data = data + return statusCacheTestBlock(slot, tx) +} + +func TestTransactionStatusCaptureSurvivesConcurrentPruneAndUnwind(t *testing.T) { + c := importedStatusCacheForTest(t) + for slot := uint64(301); slot <= 350; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + } + want, err := legacyStatusSnapshotForTest(c, 320) + require.NoError(t, err) + captured, err := c.CaptureSnapshotThrough(320) + require.NoError(t, err) + view := captured.(*transactionStatusSnapshot) + require.Len(t, view.nodes, maxTransactionStatusRoots) + require.Equal(t, uint64(21), view.nodes[0].slot) + require.Equal(t, uint64(320), view.nodes[len(view.nodes)-1].slot) + for _, node := range view.nodes { + require.Nil(t, node.parent, "capture retained excluded ancestry") + } + + var wg sync.WaitGroup + wg.Add(1) + errs := make(chan error, 1) + go func() { + defer wg.Done() + for slot := uint64(351); slot <= 750; slot++ { + if err := c.CommitBlock(captureTestBlock(slot, 1)); err != nil { + errs <- err + return + } + c.Root(slot - 20) + } + if err := c.Unwind(741); err != nil { + errs <- err + return + } + for slot := uint64(741); slot <= 755; slot++ { + if err := c.CommitBlock(captureTestBlock(slot, 2)); err != nil { + errs <- err + return + } + } + }() + for i := 0; i < 50; i++ { + got, err := captured.MarshalBinary() + if err != nil || string(want) != string(got) { + t.Errorf("captured bytes changed during replay: %v", err) + break + } + } + wg.Wait() + close(errs) + for err := range errs { + require.NoError(t, err) + } + got, err := captured.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, got) + restored, err := NewTransactionStatusCacheFromSnapshot(got) + require.NoError(t, err) + require.Equal(t, uint64(320), restored.RootedThrough()) + require.True(t, restored.CoverageComplete()) + retry := statusCacheTestBlock(321, captureTestBlock(301, 1).Transactions[0]) + require.Error(t, restored.ValidateBlock(retry), "captured ancestor was forgotten") + // A bank after the capture's through-slot must not leak into recovery. + future := statusCacheTestBlock(321, captureTestBlock(350, 1).Transactions[0]) + require.NoError(t, restored.ValidateBlock(future)) +} + +func TestTransactionStatusCapturePreservesCoverageAndOwnedBytes(t *testing.T) { + for _, complete := range []bool{false, true} { + c := newTransactionStatusCache(complete) + // Exercise metadata selection without changing its pre-existing rules. + for slot := uint64(1); slot <= 310; slot++ { + c.tip = &transactionStatusNode{slot: slot, parent: c.tip, + delta: transactionStatusDelta{solana.Hash{1}: {keyIndex: 7, keys: map[transactionStatusKey]struct{}{{byte(slot), byte(slot >> 8)}: {}}}}} + } + for _, through := range []uint64{0, 1, 299, 300, 310, 400} { + want, err := legacyStatusSnapshotForTest(c, through) + require.NoError(t, err) + view, err := c.CaptureSnapshotThrough(through) + require.NoError(t, err) + got, err := view.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, got) + got[0] ^= 0xff + again, err := view.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, again, "caller mutated the captured data through encoded bytes") + } + } + var absent *TransactionStatusCache + view, err := absent.CaptureSnapshotThrough(1) + require.NoError(t, err) + require.Nil(t, view) +} + +func TestTransactionStatusCaptureEncodingDoesNotLockLiveCache(t *testing.T) { + c := importedStatusCacheForTest(t) + require.NoError(t, c.CommitBlock(captureTestBlock(301, 1))) + view, err := c.CaptureSnapshotThrough(301) + require.NoError(t, err) + c.mu.Lock() + done := make(chan error, 1) + go func() { _, err := view.MarshalBinary(); done <- err }() + select { + case err := <-done: + c.mu.Unlock() + require.NoError(t, err) + case <-time.After(2 * time.Second): + c.mu.Unlock() + t.Fatal("checkpoint encoding waited for the live cache lock") + } +} + +func marshalTransactionStatusNodesUncached(nodes []*transactionStatusNode, rootedSinceSeed uint16, complete bool, coverageFromGenesis bool) ([]byte, error) { + var buf bytes.Buffer + buf.Write(transactionStatusSnapshotMagic[:]) + flags := byte(0) + if complete { + flags = 1 + } + if coverageFromGenesis { + flags |= 2 + } + buf.WriteByte(flags) + _ = binary.Write(&buf, binary.LittleEndian, rootedSinceSeed) + _ = binary.Write(&buf, binary.LittleEndian, uint16(len(nodes))) + for _, node := range nodes { + _ = binary.Write(&buf, binary.LittleEndian, node.slot) + nodeFlags := byte(0) + if node.hasBlockID { + nodeFlags = 1 + } + buf.WriteByte(nodeFlags) + if node.hasBlockID { + buf.Write(node.blockID[:]) + } + blockhashes := make([]solana.Hash, 0, len(node.delta)) + for blockhash := range node.delta { + blockhashes = append(blockhashes, blockhash) + } + sort.Slice(blockhashes, func(i, j int) bool { + return bytes.Compare(blockhashes[i][:], blockhashes[j][:]) < 0 + }) + _ = binary.Write(&buf, binary.LittleEndian, uint32(len(blockhashes))) + for _, blockhash := range blockhashes { + group := node.delta[blockhash] + buf.Write(blockhash[:]) + buf.WriteByte(group.keyIndex) + keys := make([]transactionStatusKey, 0, len(group.keys)) + for key := range group.keys { + keys = append(keys, key) + } + sort.Slice(keys, func(i, j int) bool { + return bytes.Compare(keys[i][:], keys[j][:]) < 0 + }) + _ = binary.Write(&buf, binary.LittleEndian, uint32(len(keys))) + for _, key := range keys { + buf.Write(key[:]) + } + } + } + return buf.Bytes(), nil +} + +func TestTransactionStatusEncodingSharedAcrossCaptureAndPrune(t *testing.T) { + for _, warmBeforePrune := range []bool{false, true} { + t.Run(fmt.Sprintf("warm=%t", warmBeforePrune), func(t *testing.T) { + c := importedStatusCacheForTest(t) + for slot := uint64(301); slot <= 305; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + } + first, err := c.CaptureSnapshotThrough(304) + require.NoError(t, err) + pinned := first.(*transactionStatusSnapshot) + want, err := legacyStatusSnapshotForTest(c, 304) + require.NoError(t, err) + if warmBeforePrune { + got, err := first.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, got) + } + // Force relinking of retained nodes after the snapshot has copied + // their headers, including the still-unencoded case. + c.Root(305) + second, err := c.CaptureSnapshotThrough(305) + require.NoError(t, err) + current := second.(*transactionStatusSnapshot) + caches := make(map[uint64]*transactionStatusNodeEncoding) + for _, node := range pinned.nodes { + caches[node.slot] = node.encodingCache() + } + for _, node := range current.nodes { + if prior := caches[node.slot]; prior != nil { + require.Same(t, prior, node.encodingCache()) + } + } + var wg sync.WaitGroup + for i := 0; i < 8; i++ { + wg.Add(1) + go func() { + defer wg.Done() + got, err := first.MarshalBinary() + assert.NoError(t, err) + assert.Equal(t, want, got) + // Mutate the node body as well as the header; neither may + // alias the memoized node data or another caller's result. + clear(got) + }() + } + wg.Wait() + currentWant, err := legacyStatusSnapshotForTest(c, 305) + require.NoError(t, err) + got, err := second.MarshalBinary() + require.NoError(t, err) + require.Equal(t, currentWant, got) + restored, err := NewTransactionStatusCacheFromSnapshot(got) + require.NoError(t, err) + roundTrip, err := restored.SnapshotThrough(305) + require.NoError(t, err) + require.Equal(t, got, roundTrip) + }) + } +} + +func TestTransactionStatusEncodingMatchesOriginalWireFormat(t *testing.T) { + // Deliberately unsorted groups/keys, nonzero offsets, empty deltas and + // mixed block-ID presence exercise every independently cached field. + nodes := []*transactionStatusNode{ + {slot: 3, hasBlockID: true, blockID: solana.Hash{9}, delta: transactionStatusDelta{ + solana.Hash{7}: {keyIndex: 11, keys: map[transactionStatusKey]struct{}{{8}: {}, {1}: {}, {4}: {}}}, + solana.Hash{1}: {keyIndex: 2, keys: map[transactionStatusKey]struct{}{{9}: {}, {2}: {}}}, + }}, + {slot: 5}, + {slot: 8, delta: transactionStatusDelta{solana.Hash{3}: {keyIndex: 0, keys: map[transactionStatusKey]struct{}{}}}}, + } + for _, complete := range []bool{false, true} { + for _, genesis := range []bool{false, true} { + want, err := marshalTransactionStatusNodesUncached(nodes, 3, complete, genesis) + require.NoError(t, err) + got, err := marshalTransactionStatusNodes(nodes, 3, complete, genesis) + require.NoError(t, err) + require.Equal(t, want, got) + } + } +} + +func TestTransactionStatusEncodingRandomizedByteIdentity(t *testing.T) { + rng := rand.New(rand.NewSource(20260916)) + for trial := 0; trial < 200; trial++ { + nodes := make([]*transactionStatusNode, rng.Intn(13)) + var slot uint64 + for i := range nodes { + slot += uint64(1 + rng.Intn(5)) + node := &transactionStatusNode{slot: slot, hasBlockID: rng.Intn(2) == 1, delta: make(transactionStatusDelta)} + _, _ = rng.Read(node.blockID[:]) + for g := rng.Intn(8); g > 0; g-- { + var hash solana.Hash + _, _ = rng.Read(hash[:]) + group := &transactionStatusGroup{keyIndex: uint8(rng.Intn(int(txstatus.MaxCachedKeyIndex) + 1)), keys: make(map[transactionStatusKey]struct{})} + for k := rng.Intn(13); k > 0; k-- { + var key transactionStatusKey + _, _ = rng.Read(key[:]) + group.keys[key] = struct{}{} + } + node.delta[hash] = group + } + nodes[i] = node + } + rooted := uint16(rng.Intn(301)) + complete, genesis := rng.Intn(2) == 1, rng.Intn(2) == 1 + want, err := marshalTransactionStatusNodesUncached(nodes, rooted, complete, genesis) + require.NoError(t, err) + for pass := 0; pass < 2; pass++ { + got, err := marshalTransactionStatusNodes(nodes, rooted, complete, genesis) + require.NoError(t, err) + require.Equal(t, want, got, "trial=%d pass=%d", trial, pass) + } + } +} diff --git a/pkg/replay/transaction_status_expiry_test.go b/pkg/replay/transaction_status_expiry_test.go new file mode 100644 index 000000000..84a11cd28 --- /dev/null +++ b/pkg/replay/transaction_status_expiry_test.go @@ -0,0 +1,129 @@ +package replay + +import ( + "encoding/binary" + "fmt" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" + "math/rand" + "testing" +) + +func TestTransactionStatusBatchExpiryMatchesPerKeyRemoval(t *testing.T) { + for seed := int64(0); seed < 100; seed++ { + rng := rand.New(rand.NewSource(seed)) + fast, ref := NewTransactionStatusCache(), NewTransactionStatusCache() + var nodes []*transactionStatusNode + for slot := 0; slot < 40; slot++ { + d := make(transactionStatusDelta) + for j := 0; j < 6; j++ { + h := solana.Hash{byte(rng.Intn(12))} + g := &transactionStatusGroup{keyIndex: h[0], keys: make(map[transactionStatusKey]struct{})} + for k := 0; k < rng.Intn(30); k++ { + g.keys[transactionStatusKey{byte(rng.Intn(40))}] = struct{}{} + } + d[h] = g + } + nodes = append(nodes, &transactionStatusNode{slot: uint64(slot), delta: d}) + require.NoError(t, fast.addDeltaVisibleLocked(d)) + require.NoError(t, ref.addDeltaVisibleLocked(d)) + } + cut := 1 + rng.Intn(len(nodes)-1) + fast.expireVisibleLocked(nodes[:cut], nodes[cut:]) + for _, n := range nodes[:cut] { + ref.removeDeltaVisibleLocked(n.delta) + } + require.Equal(t, ref.visible, fast.visible, "seed %d", seed) + for i := len(nodes) - 1; i >= cut; i-- { + fast.removeDeltaVisibleLocked(nodes[i].delta) + ref.removeDeltaVisibleLocked(nodes[i].delta) + } + require.Equal(t, ref.visible, fast.visible, "unwind seed %d", seed) + } +} + +func TestTransactionStatusBatchExpiryPinnedViewsAndSnapshot(t *testing.T) { + c := NewTransactionStatusCache() + old := statusCacheTestTransaction(1, 1, 1) + keep := statusCacheTestTransaction(2, 2, 2) + require.NoError(t, c.CommitBlock(statusCacheTestBlock(1, old))) + for slot := uint64(2); slot <= maxTransactionStatusRoots+1; slot++ { + blk := statusCacheTestBlock(slot) + if slot == maxTransactionStatusRoots+1 { + blk.Transactions = append(blk.Transactions, keep) + } + require.NoError(t, c.CommitBlock(blk)) + } + pinned := c.View() + snapshot, err := c.CaptureSnapshotThrough(maxTransactionStatusRoots + 1) + require.NoError(t, err) + before, err := snapshot.MarshalBinary() + require.NoError(t, err) + c.coverageComplete = false // Exercise completion once 300 banks become rooted. + c.Root(maxTransactionStatusRoots + 1) + after, err := snapshot.MarshalBinary() + require.NoError(t, err) + require.Equal(t, before, after) + found, err := pinned.ContainsTransaction(old) + require.NoError(t, err) + require.True(t, found) + found, err = c.View().ContainsTransaction(old) + require.NoError(t, err) + require.False(t, found) + require.NoError(t, c.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, old))) + require.Error(t, c.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, keep))) + blob, err := c.SnapshotThrough(maxTransactionStatusRoots + 1) + require.NoError(t, err) + restored, err := NewTransactionStatusCacheFromSnapshot(blob) + require.NoError(t, err) + require.NoError(t, restored.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, old))) + require.Error(t, restored.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, keep))) + require.Error(t, c.Unwind(maxTransactionStatusRoots+1)) +} + +func BenchmarkTransactionStatusBatchExpiry(b *testing.B) { + for _, shape := range []string{"four-bank-groups", "one-expired-group", "crossing-group"} { + for _, legacy := range []bool{true, false} { + b.Run(fmt.Sprintf("%s/legacy=%t", shape, legacy), func(b *testing.B) { + for i := 0; i < b.N; i++ { + b.StopTimer() + c := NewTransactionStatusCache() + var expired, retained []*transactionStatusNode + for slot := 0; slot < 129; slot++ { + var h solana.Hash + if shape == "four-bank-groups" || (shape == "one-expired-group" && slot == 128) { + binary.LittleEndian.PutUint64(h[:], uint64(slot/4+1)) + } + g := &transactionStatusGroup{keys: make(map[transactionStatusKey]struct{})} + for k := 0; k < 33760; k++ { + var key transactionStatusKey + binary.LittleEndian.PutUint64(key[:], uint64(slot*33760+k)) + g.keys[key] = struct{}{} + } + n := &transactionStatusNode{delta: transactionStatusDelta{h: g}} + if err := c.addDeltaVisibleLocked(n.delta); err != nil { + b.Fatal(err) + } + if slot < 128 { + expired = append(expired, n) + } else { + retained = append(retained, n) + } + } + b.StartTimer() + if legacy { + for _, n := range expired { + c.removeDeltaVisibleLocked(n.delta) + } + } else { + c.expireVisibleLocked(expired, retained) + } + b.StopTimer() + if len(c.visible) != 1 { + b.Fatal("retained group missing") + } + } + }) + } + } +} diff --git a/pkg/replay/transaction_status_overlap_benchmark_test.go b/pkg/replay/transaction_status_overlap_benchmark_test.go new file mode 100644 index 000000000..0da5f739f --- /dev/null +++ b/pkg/replay/transaction_status_overlap_benchmark_test.go @@ -0,0 +1,110 @@ +package replay + +import ( + "fmt" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" +) + +// This controlled workload runs 4,096 executions of the transfer fixture while +// preparing 33,760 independent status keys. It measures scheduling/GC contention, +// not full replay: no accounts are committed, and the status fixture differs +// from the repeated transfer fixture. Run with -cpu=1,2 to compare contention +// without and with a spare execution thread. Check live replay separately. +func BenchmarkTransactionStatusExecutionOverlap(tb *testing.B) { + for _, mode := range []string{"legacy", "sized", "overlap"} { + tb.Run(mode, func(tb *testing.B) { + slotCtx, cleanup := newCommitTestSlotCtx() + defer cleanup() + tx, err := solana.TransactionFromBytes(txfixture.MustSignedTransferWire(0)) + if err != nil { + tb.Fatal(err) + } + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(statusCacheTestBlock(10)); err != nil { + tb.Fatal(err) + } + blk := statusCacheTestBlock(11, benchmarkUniqueTransactions(33760)...) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + tb.Fatal(err) + } + var execution, commit time.Duration + tb.ReportAllocs() + tb.ResetTimer() + for range tb.N { + var p *transactionStatusPreparation + if mode == "overlap" { + p = cache.startStatusPreparation(plan) + } + start := time.Now() + for range 4096 { + output := LoadAndExecuteTransaction(LoadAndExecuteTransactionInput{SlotCtx: slotCtx, Transaction: tx, LeanResult: true}) + if output.ProcessingResult.TransactionError != nil { + tb.Fatal(output.ProcessingResult.TransactionError) + } + } + execution += time.Since(start) + start = time.Now() + switch mode { + case "legacy": + err = cache.legacyCommitStatusForBenchmark(blk, plan) + case "sized": + err = cache.commitBlockWithPlan(blk, plan) + case "overlap": + err = cache.commitBlockWithPreparedDelta(blk, plan, p.wait()) + } + commit += time.Since(start) + if err != nil { + tb.Fatal(err) + } + tb.StopTimer() + if err := cache.Unwind(11); err != nil { + tb.Fatal(err) + } + tb.StartTimer() + } + tb.StopTimer() + tb.ReportMetric(float64(execution.Nanoseconds())/float64(tb.N), "execution-ns/op") + tb.ReportMetric(float64(commit.Nanoseconds())/float64(tb.N), "commit-with-wait-ns/op") + }) + } +} + +func BenchmarkTransactionStatusSmallPublication(tb *testing.B) { + for _, count := range []int{0, 1, 32} { + for _, mode := range []string{"legacy", "prepared_total"} { + tb.Run(fmt.Sprintf("txs_%d/%s", count, mode), func(tb *testing.B) { + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(statusCacheTestBlock(10)); err != nil { + tb.Fatal(err) + } + blk := statusCacheTestBlock(11, benchmarkUniqueTransactions(count)...) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + tb.Fatal(err) + } + tb.ReportAllocs() + tb.ResetTimer() + for range tb.N { + if mode == "legacy" { + err = cache.legacyCommitStatusForBenchmark(blk, plan) + } else { + err = cache.commitBlockWithPreparedDelta(blk, plan, cache.startStatusPreparation(plan).wait()) + } + if err != nil { + tb.Fatal(err) + } + // Include unwind equally in this small-work benchmark, avoiding + // timer start/stop overhead around microsecond operations. + if err := cache.Unwind(11); err != nil { + tb.Fatal(err) + } + } + }) + } + } +} diff --git a/pkg/replay/transaction_status_plan_binding_test.go b/pkg/replay/transaction_status_plan_binding_test.go index 30b456d3a..faf0b2540 100644 --- a/pkg/replay/transaction_status_plan_binding_test.go +++ b/pkg/replay/transaction_status_plan_binding_test.go @@ -15,9 +15,10 @@ func TestPreparedCommitRejectsTransactionReplacement(t *testing.T) { if err != nil { t.Fatal(err) } + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) candidate.Transactions[0] = statusCacheTestTransaction(4, 5, 6) - err = cache.commitBlockWithPlan(candidate, plan) + err = cache.commitBlockWithPreparedDelta(candidate, plan, prepared) if err == nil || err.Error() != "prepared transaction message identities do not match block" { t.Fatalf("commit error = %v, want prepared-plan binding failure", err) } diff --git a/pkg/replay/transaction_status_prepared_test.go b/pkg/replay/transaction_status_prepared_test.go index 44a42ca83..0172fffa6 100644 --- a/pkg/replay/transaction_status_prepared_test.go +++ b/pkg/replay/transaction_status_prepared_test.go @@ -25,7 +25,9 @@ func TestPreparedCommitRechecksAncestorAfterForkSwitch(t *testing.T) { candidate := statusCacheTestBlock(12, retried, unique) plan, err := planBlockTransactionExecution(candidate) requireNoError(err) - requireNoError(cache.validateBlockWithPlan(candidate, plan)) + validation, err := cache.validateBlockForPublication(candidate, plan) + requireNoError(err) + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) requireNoError(cache.Unwind(11)) replacement := statusCacheTestBlock( @@ -34,7 +36,7 @@ func TestPreparedCommitRechecksAncestorAfterForkSwitch(t *testing.T) { ) requireNoError(cache.CommitBlock(replacement)) - err = cache.commitBlockWithPlan(candidate, plan) + err = cache.commitBlockWithValidation(candidate, plan, prepared, validation) var ancestorErr *AncestorAlreadyProcessedTransactionMessagesError if !errors.As(err, &ancestorErr) { t.Fatalf("prepared commit error = %v, want ancestor AlreadyProcessed", err) @@ -76,22 +78,26 @@ func TestConcurrentPreparedSiblingCommitsPublishExactlyOne(t *testing.T) { if err != nil { t.Fatal(err) } - if err := cache.validateBlockWithPlan(left, leftPlan); err != nil { + leftValidation, err := cache.validateBlockForPublication(left, leftPlan) + if err != nil { t.Fatalf("prevalidate left sibling: %v", err) } - if err := cache.validateBlockWithPlan(right, rightPlan); err != nil { + rightValidation, err := cache.validateBlockForPublication(right, rightPlan) + if err != nil { t.Fatalf("prevalidate right sibling: %v", err) } + leftPrepared := cache.prepareTransactionStatusDelta(leftPlan.messageIdentities) + rightPrepared := cache.prepareTransactionStatusDelta(rightPlan.messageIdentities) start := make(chan struct{}) results := make(chan error, 2) go func() { <-start - results <- cache.commitBlockWithPlan(left, leftPlan) + results <- cache.commitBlockWithValidation(left, leftPlan, leftPrepared, leftValidation) }() go func() { <-start - results <- cache.commitBlockWithPlan(right, rightPlan) + results <- cache.commitBlockWithValidation(right, rightPlan, rightPrepared, rightValidation) }() close(start) diff --git a/pkg/replay/transaction_status_publication.go b/pkg/replay/transaction_status_publication.go new file mode 100644 index 000000000..e821071ad --- /dev/null +++ b/pkg/replay/transaction_status_publication.go @@ -0,0 +1,83 @@ +package replay + +import ( + "runtime" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" +) + +// Preparation owns private, immutable maps. It never publishes a status or +// authorizes a bank: commit still checks coverage, lineage, and duplicates. +type preparedTransactionStatusDelta struct { + identities *b.PreparedTransactionMessageIdentities + delta transactionStatusDelta +} + +type transactionStatusPreparation struct { + done chan struct{} + prepared *preparedTransactionStatusDelta + duration time.Duration +} + +// Replay joins this task on every exit, including rejected banks. It only reads +// the immutable identities, so account loading and ALT resolution can proceed. +func (c *TransactionStatusCache) startStatusPreparation(plan blockTransactionExecutionPlan) *transactionStatusPreparation { + // Small-block measurements show dispatch/join costs as much as the work. + // With one Go execution thread preparation cannot overlap execution at all. + if plan.messageIdentities.Len() <= 32 || runtime.GOMAXPROCS(0) == 1 { + return nil + } + p := &transactionStatusPreparation{done: make(chan struct{})} + go func() { + defer close(p.done) + start := time.Now() + p.prepared = c.prepareTransactionStatusDelta(plan.messageIdentities) + p.duration = time.Since(start) + }() + return p +} + +func (p *transactionStatusPreparation) wait() *preparedTransactionStatusDelta { + if p == nil { + return nil + } + <-p.done + return p.prepared +} + +func countTransactionStatusGroups(identities *b.PreparedTransactionMessageIdentities) map[solana.Hash]int { + counts := make(map[solana.Hash]int) + for i := 0; i < identities.Len(); i++ { + counts[identities.Identity(i).RecentBlockhash]++ + } + return counts +} + +func buildTransactionStatusDelta(identities *b.PreparedTransactionMessageIdentities, counts map[solana.Hash]int, indexes map[solana.Hash]uint8) transactionStatusDelta { + delta := make(transactionStatusDelta, len(counts)) + for blockhash, count := range counts { + delta[blockhash] = &transactionStatusGroup{keyIndex: indexes[blockhash], keys: make(map[transactionStatusKey]struct{}, count)} + } + for i := 0; i < identities.Len(); i++ { + identity := identities.Identity(i) + group := delta[identity.RecentBlockhash] + group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + } + return delta +} + +func (c *TransactionStatusCache) prepareTransactionStatusDelta(identities *b.PreparedTransactionMessageIdentities) *preparedTransactionStatusDelta { + counts := countTransactionStatusGroups(identities) + indexes := make(map[solana.Hash]uint8, len(counts)) + // Copy offsets only, never share mutable visible maps with the worker. + c.mu.RLock() + for blockhash := range counts { + if visible := c.visible[blockhash]; visible != nil { + indexes[blockhash] = visible.keyIndex + } + } + c.mu.RUnlock() + return &preparedTransactionStatusDelta{identities: identities, delta: buildTransactionStatusDelta(identities, counts, indexes)} +} diff --git a/pkg/replay/transaction_status_publication_benchmark_test.go b/pkg/replay/transaction_status_publication_benchmark_test.go new file mode 100644 index 000000000..2151ef330 --- /dev/null +++ b/pkg/replay/transaction_status_publication_benchmark_test.go @@ -0,0 +1,169 @@ +package replay + +import ( + "encoding/binary" + "errors" + "fmt" + "testing" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" +) + +func (c *TransactionStatusCache) legacyCommitStatusForBenchmark(block *b.Block, plan blockTransactionExecutionPlan) error { + if block == nil || plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { + return errors.New("prepared transaction message identities do not match block") + } + c.mu.Lock() + defer c.mu.Unlock() + if !c.coverageComplete { + return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} + } + // Parent lineage and ancestor status are mutable, so both remain under the + // publication lock even when hashing and same-bank deduplication happened + // earlier. This keeps commit safe across a concurrent branch transition. + if err := c.validateParentLocked(block); err != nil { + return err + } + if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { + return err + } + + delta := make(transactionStatusDelta) + for index := 0; index < plan.messageIdentities.Len(); index++ { + identity := plan.messageIdentities.Identity(index) + blockhash := identity.RecentBlockhash + group := delta[blockhash] + if group == nil { + keyIndex := uint8(0) + if visible := c.visible[blockhash]; visible != nil { + keyIndex = visible.keyIndex + } + group = &transactionStatusGroup{ + keyIndex: keyIndex, + keys: make(map[transactionStatusKey]struct{}), + } + delta[blockhash] = group + } + group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + } + + if err := c.legacyAddStatusForBenchmark(delta); err != nil { + return err + } + c.tip = &transactionStatusNode{ + slot: block.Slot, + blockID: solana.Hash(block.AlpenglowBlockID), + hasBlockID: block.HasAlpenglowBlockID, + parent: c.tip, + delta: delta, + } + return nil +} + +// Frozen production commit algorithm before publication optimization. This is +// an independent baseline, including its original visible-index allocation. +// Benchmark fixtures never reuse validation receipts on this path. This frozen +// helper deliberately omits validation-version bumps and must not be used by +// production callers or copied as a model for mutating the live cache. +func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionStatusDelta) error { + for blockhash, deltaGroup := range delta { + if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { + return fmt.Errorf("transaction status blockhash %s uses inconsistent key indexes %d and %d", + blockhash, group.keyIndex, deltaGroup.keyIndex) + } + } + for blockhash, deltaGroup := range delta { + group := c.visible[blockhash] + if group == nil { + group = &visibleTransactionStatusGroup{ + keyIndex: deltaGroup.keyIndex, + keys: make(map[transactionStatusKey]uint16), + } + c.visible[blockhash] = group + } + for key := range deltaGroup.keys { + group.keys[key]++ + } + } + return nil +} + +// BenchmarkTransactionStatusPublication times only status publication, with +// prepared message identities. No execution, disk I/O, signing or networking. +// prepared_commit excludes delta preparation; prepared_total includes it and +// goroutine dispatch/join, with no execution overlap. Neither measures replay. +// validated_commit also excludes the successful pre-execution ancestor scan. +// invalidated_commit roots between validation and commit, forcing a full recheck. +// Each iteration restores the same ancestor contents; existing maps retain +// steady-state capacity. Fixture creation, seeding and unwind are not timed. +func BenchmarkTransactionStatusPublication(tb *testing.B) { + const count = 33760 + for _, groups := range []int{1, 4} { + for _, existing := range []bool{false, true} { + tb.Run(fmt.Sprintf("groups_%d/existing_%t", groups, existing), func(tb *testing.B) { + txs := benchmarkUniqueTransactions(count * 2) + for i, tx := range txs { + binary.LittleEndian.PutUint32(tx.Message.RecentBlockhash[:], uint32(i%groups+1)) + } + parent := statusCacheTestBlock(10, txs[:count]...) + if !existing { + parent.Transactions = nil + } + blk := statusCacheTestBlock(11, txs[count:]...) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + tb.Fatal(err) + } + for _, name := range []string{"legacy", "sized", "prepared_total", "prepared_commit", "validated_commit", "invalidated_commit"} { + tb.Run(name, func(tb *testing.B) { + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(parent); err != nil { + tb.Fatal(err) + } + tb.ReportAllocs() + tb.ResetTimer() + for range tb.N { + var err error + if name == "prepared_commit" || name == "validated_commit" || name == "invalidated_commit" { + tb.StopTimer() + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) + validation, validationErr := cache.validateBlockForPublication(blk, plan) + if validationErr != nil { + tb.Fatal(validationErr) + } + if name == "invalidated_commit" { + cache.Root(10) + } + tb.StartTimer() + if name == "prepared_commit" { + err = cache.commitBlockWithPreparedDelta(blk, plan, prepared) + } else { + err = cache.commitBlockWithValidation(blk, plan, prepared, validation) + } + } else if name == "prepared_total" { + prepared := cache.startStatusPreparation(plan).wait() + err = cache.commitBlockWithPreparedDelta(blk, plan, prepared) + } else if name == "legacy" { + err = cache.legacyCommitStatusForBenchmark(blk, plan) + } else { + err = cache.commitBlockWithPlan(blk, plan) + } + if err != nil { + tb.Fatal(err) + } + tb.StopTimer() + if got := cache.tip.slot; got != 11 { + tb.Fatalf("tip=%d", got) + } + if err := cache.Unwind(11); err != nil { + tb.Fatal(err) + } + tb.StartTimer() + } + }) + } + }) + } + } +} diff --git a/pkg/replay/transaction_status_publication_test.go b/pkg/replay/transaction_status_publication_test.go new file mode 100644 index 000000000..3b9b92885 --- /dev/null +++ b/pkg/replay/transaction_status_publication_test.go @@ -0,0 +1,154 @@ +package replay + +import ( + "fmt" + "runtime" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/stretchr/testify/require" +) + +func TestPreparedStatusDeltaRebindsSnapshotOffsets(t *testing.T) { + for _, from := range []uint64{0, 7, txstatus.MaxCachedKeyIndex} { + for _, to := range []uint64{0, 7, txstatus.MaxCachedKeyIndex} { + t.Run(fmt.Sprintf("%d_to_%d", from, to), func(t *testing.T) { + ancestor := statusCacheTestTransaction(1, 2, 3) + seed := func(offset uint64) *TransactionStatusCache { + cache, err := NewTransactionStatusCacheFromAgaveSnapshot([]txstatus.SnapshotSlotDelta{ + {Slot: 0, IsRoot: true, Statuses: []txstatus.SnapshotStatus{snapshotStatusCacheStatusForTx(t, ancestor, offset)}}, + }, 0) + require.NoError(t, err) + return cache + } + candidate := statusCacheTestBlock(1, statusCacheTestTransaction(1, 4, 5)) + plan, err := planBlockTransactionExecution(candidate) + require.NoError(t, err) + prepared := seed(from).prepareTransactionStatusDelta(plan.messageIdentities) + cache := seed(to) + pinned := cache.View() + require.NoError(t, cache.commitBlockWithPreparedDelta(candidate, plan, prepared)) + require.Equal(t, uint8(to), cache.tip.delta[candidate.Transactions[0].Message.RecentBlockhash].keyIndex) + found, err := cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.True(t, found) + found, err = pinned.ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + blob, err := cache.SnapshotThrough(1) + require.NoError(t, err) + restored, err := NewTransactionStatusCacheFromSnapshot(blob) + require.NoError(t, err) + found, err = restored.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.True(t, found) + require.NoError(t, cache.Unwind(1)) + found, err = cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + found, err = cache.View().ContainsTransaction(ancestor) + require.NoError(t, err) + require.True(t, found) + }) + } + } +} + +func TestPreparedStatusDeltaDoesNotPublishUntilCommit(t *testing.T) { + prior := runtime.GOMAXPROCS(2) + defer runtime.GOMAXPROCS(prior) + cache := NewTransactionStatusCache() + require.NoError(t, cache.CommitBlock(statusCacheTestBlock(10))) + candidate := statusCacheTestBlock(11, benchmarkUniqueTransactions(33760)...) + plan, err := planBlockTransactionExecution(candidate) + require.NoError(t, err) + p := cache.startStatusPreparation(plan) + // A rejected bank joins and discards the prepared maps. Waiting is also + // idempotent for the normal commit followed by ProcessBlock's deferred join. + require.Same(t, p.wait(), p.wait()) + found, err := cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + require.Equal(t, uint64(10), cache.tip.slot) + cache.mu.Lock() + cache.coverageComplete = false + cache.mu.Unlock() + var incomplete *IncompleteTransactionStatusCoverageError + require.ErrorAs(t, cache.commitBlockWithPreparedDelta(candidate, plan, p.wait()), &incomplete) + require.Equal(t, uint64(10), cache.tip.slot) +} + +func TestPreparedStatusDeltaRejectsWrongPlanWithoutPublishingIt(t *testing.T) { + cache := NewTransactionStatusCache() + require.NoError(t, cache.CommitBlock(statusCacheTestBlock(10))) + left := statusCacheTestBlock(11, statusCacheTestTransaction(1, 2, 3)) + right := statusCacheTestBlock(11, statusCacheTestTransaction(1, 4, 5)) + leftPlan, err := planBlockTransactionExecution(left) + require.NoError(t, err) + rightPlan, err := planBlockTransactionExecution(right) + require.NoError(t, err) + prepared := cache.prepareTransactionStatusDelta(leftPlan.messageIdentities) + // A mismatched prepared delta falls back to the actual block's identities. + require.NoError(t, cache.commitBlockWithPreparedDelta(right, rightPlan, prepared)) + found, err := cache.View().ContainsTransaction(left.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + found, err = cache.View().ContainsTransaction(right.Transactions[0]) + require.NoError(t, err) + require.True(t, found) +} + +func TestPreparedStatusDeltaEmptyBlock(t *testing.T) { + cache := NewTransactionStatusCache() + block := statusCacheTestBlock(10) + plan, err := planBlockTransactionExecution(block) + require.NoError(t, err) + p := cache.startStatusPreparation(plan) + require.Nil(t, p, "empty bank must not queue background work") + require.NoError(t, cache.commitBlockWithPreparedDelta(block, plan, p.wait())) +} + +func TestPreparedStatusDeltaScheduling(t *testing.T) { + for _, threads := range []int{1, 2} { + for _, count := range []int{1, 32, 33} { + t.Run(fmt.Sprintf("threads_%d/txs_%d", threads, count), func(t *testing.T) { + previous := runtime.GOMAXPROCS(threads) + defer runtime.GOMAXPROCS(previous) + cache := NewTransactionStatusCache() + block := statusCacheTestBlock(10, benchmarkUniqueTransactions(count)...) + plan, err := planBlockTransactionExecution(block) + require.NoError(t, err) + p := cache.startStatusPreparation(plan) + defer p.wait() + require.Equal(t, threads > 1 && count > 32, p != nil) + require.NoError(t, cache.commitBlockWithPreparedDelta(block, plan, p.wait())) + for _, tx := range block.Transactions { + found, err := cache.View().ContainsTransaction(tx) + require.NoError(t, err) + require.True(t, found) + } + }) + } + } +} + +func TestPreparedStatusDeltaRebindsMissingSnapshotGroup(t *testing.T) { + ancestor := statusCacheTestTransaction(1, 2, 3) + seed, err := NewTransactionStatusCacheFromAgaveSnapshot([]txstatus.SnapshotSlotDelta{ + {Slot: 0, IsRoot: true, Statuses: []txstatus.SnapshotStatus{snapshotStatusCacheStatusForTx(t, ancestor, 7)}}, + }, 0) + require.NoError(t, err) + candidate := statusCacheTestBlock(1, statusCacheTestTransaction(1, 4, 5)) + plan, err := planBlockTransactionExecution(candidate) + require.NoError(t, err) + prepared := seed.prepareTransactionStatusDelta(plan.messageIdentities) + // Model restore/branch replacement removing the group after preparation. + cache := NewTransactionStatusCache() + require.NoError(t, cache.commitBlockWithPreparedDelta(candidate, plan, prepared)) + inline := NewTransactionStatusCache() + require.NoError(t, inline.commitBlockWithPlan(candidate, plan)) + require.Equal(t, inline.tip.delta, cache.tip.delta) + found, err := cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.True(t, found) +} diff --git a/pkg/replay/transaction_status_validation.go b/pkg/replay/transaction_status_validation.go new file mode 100644 index 000000000..4413b8b8f --- /dev/null +++ b/pkg/replay/transaction_status_validation.go @@ -0,0 +1,37 @@ +package replay + +import ( + "math" + + b "github.com/Overclock-Validator/mithril/pkg/block" +) + +// transactionStatusValidation records a successful ancestor scan under cache.mu. +// It authorizes skipping only that scan, never the coverage, parent or exact +// block-identity checks. The receipt is private, bound to one cache instance and +// one immutable identity set, and checked under the publication lock. +// +// Mutating the visible index, binding the tip, rooting/pruning or restoring +// invalidates earlier receipts. In particular, committing and then unwinding to +// the same tip cannot resurrect one. Snapshot/Agave constructors create a new +// cache instance; this receipt is neither persisted nor usable after recovery. +// This optimization changes no crash-recovery or durable-checkpoint guarantee. +type transactionStatusValidation struct { + cache *TransactionStatusCache + identities *b.PreparedTransactionMessageIdentities + version uint64 +} + +func (v transactionStatusValidation) reusableForLocked(c *TransactionStatusCache, identities *b.PreparedTransactionMessageIdentities) bool { + return v.cache == c && v.identities == identities && + v.version == c.validationVersion && c.validationVersion != math.MaxUint64 +} + +// invalidateValidationLocked requires exclusive access (mu, or an unpublished +// constructor). Saturation permanently disables reuse instead of wrapping into +// an old generation. Even empty commits invalidate, because they change lineage. +func (c *TransactionStatusCache) invalidateValidationLocked() { + if c.validationVersion != math.MaxUint64 { + c.validationVersion++ + } +} diff --git a/pkg/replay/transaction_status_validation_test.go b/pkg/replay/transaction_status_validation_test.go new file mode 100644 index 000000000..9a1346125 --- /dev/null +++ b/pkg/replay/transaction_status_validation_test.go @@ -0,0 +1,151 @@ +package replay + +import ( + "errors" + "math" + "testing" + + "github.com/gagliardetto/solana-go" +) + +func TestStatusValidationInvalidation(t *testing.T) { + for _, action := range []string{"commit", "unwind", "round_trip", "root", "bind", "restore"} { + t.Run(action, func(t *testing.T) { + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(statusCacheTestBlock(10)); err != nil { + t.Fatal(err) + } + blk := statusCacheTestBlock(11, statusCacheTestTransaction(1, 2, 3)) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + if !receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("unchanged receipt not reusable") + } + switch action { + case "commit", "round_trip": + err = cache.CommitBlock(statusCacheTestBlock(11)) + if err == nil && action == "round_trip" { + err = cache.Unwind(11) + } + case "unwind": + err = cache.Unwind(10) + case "root": + cache.Root(10) + case "bind": + err = cache.BindTipBlockID(10, solana.Hash{1}) + case "restore": + var data []byte + data, err = cache.SnapshotThrough(10) + if err == nil { + cache, err = NewTransactionStatusCacheFromSnapshot(data) + } + } + if err != nil { + t.Fatal(err) + } + if receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("receipt survived " + action) + } + }) + } +} + +func TestStatusValidationCannotCrossCacheOrIdentity(t *testing.T) { + good := NewTransactionStatusCache() + bad := NewTransactionStatusCache() + tx := statusCacheTestTransaction(1, 2, 3) + if err := good.CommitBlock(statusCacheTestBlock(10)); err != nil { + t.Fatal(err) + } + if err := bad.CommitBlock(statusCacheTestBlock(10, tx)); err != nil { + t.Fatal(err) + } + blk := statusCacheTestBlock(11, tx) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := good.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + // Both caches have the same version and parent slot, but different contents. + if good.validationVersion != bad.validationVersion { + t.Fatal("fixture must have equal versions") + } + prepared := bad.prepareTransactionStatusDelta(plan.messageIdentities) + var already *AncestorAlreadyProcessedTransactionMessagesError + if err := bad.commitBlockWithValidation(blk, plan, prepared, receipt); !errors.As(err, &already) { + t.Fatalf("foreign cache: %v", err) + } + + unique := statusCacheTestBlock(11, statusCacheTestTransaction(4, 5, 6)) + uniquePlan, err := planBlockTransactionExecution(unique) + if err != nil { + t.Fatal(err) + } + receipt, err = bad.validateBlockForPublication(unique, uniquePlan) + if err != nil { + t.Fatal(err) + } + if err := bad.commitBlockWithValidation(blk, plan, prepared, receipt); !errors.As(err, &already) { + t.Fatalf("foreign identities: %v", err) + } + failed, err := bad.validateBlockForPublication(blk, plan) + if err == nil || failed.cache != nil { + t.Fatalf("failed validation returned a receipt: %+v, %v", failed, err) + } +} + +func TestStatusValidationSaturation(t *testing.T) { + cache := NewTransactionStatusCache() + cache.validationVersion = math.MaxUint64 - 1 + blk := statusCacheTestBlock(1) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + cache.Root(0) + cache.Root(0) + if cache.validationVersion != math.MaxUint64 || receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("generation wrapped or old receipt reusable") + } + receipt, err = cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + if receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("saturated cache allowed reuse") + } + if err := cache.commitBlockWithValidation(blk, plan, nil, receipt); err != nil { + t.Fatal(err) + } +} + +func TestStatusValidationStillChecksBlockBinding(t *testing.T) { + cache := NewTransactionStatusCache() + blk := statusCacheTestBlock(1, statusCacheTestTransaction(1, 2, 3)) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) + blk.Transactions[0] = statusCacheTestTransaction(4, 5, 6) + if err := cache.commitBlockWithValidation(blk, plan, prepared, receipt); err == nil { + t.Fatal("replaced transaction accepted") + } +} diff --git a/pkg/rewards/alpenglow_rewards_test.go b/pkg/rewards/alpenglow_rewards_test.go index 76136886d..ae9d451bb 100644 --- a/pkg/rewards/alpenglow_rewards_test.go +++ b/pkg/rewards/alpenglow_rewards_test.go @@ -116,6 +116,59 @@ func TestAlpenglowEarnedPointsAreNotCreditsOnly(t *testing.T) { )) } +func TestAlpenglowSkippedRewardCreditsRespectStakeActivation(t *testing.T) { + votePubkey := solana.PublicKey{1} + voteState := &sealevel.VoteStateVersions{ + Type: sealevel.VoteStateVersionV4, + V4: sealevel.VoteState4{EpochCredits: []sealevel.EpochCredits{{ + Epoch: 115, Credits: 2_000, PrevCredits: 1_000, + }}}, + } + mode := RewardCalculationMode{ + FullAlpenglow: true, + RewardEpochDelegatedStakes: map[solana.PublicKey]uint64{votePubkey: 1_000_000}, + } + for _, tc := range []struct { + name string + activation, deactivation uint64 + advance bool + }{ + {"fully cooled in rewarded epoch", 103, 114, false}, + {"not yet activating", 116, math.MaxUint64, false}, + {"activating in rewarded epoch", 115, math.MaxUint64, true}, + {"effective fractional reward", 103, math.MaxUint64, true}, + {"still cooling in rewarded epoch", 103, 115, true}, + } { + t.Run(tc.name, func(t *testing.T) { + delegation := &sealevel.Delegation{ + VoterPubkey: votePubkey, StakeLamports: 1, + ActivationEpoch: tc.activation, DeactivationEpoch: tc.deactivation, + CreditsObserved: 1_000, + } + pcs := calculateStakePointsAndCredits(solana.PublicKey{}, &sealevel.SysvarStakeHistory{}, + delegation, voteState, nil, 115, mode) + require.True(t, pcs.Points.Eq(wide.Uint128{})) + require.Equal(t, uint64(2_000), pcs.NewCreditsObserved) + require.Equal(t, tc.advance, shouldForceCreditsOnly(pcs, 1, tc.activation, 115, 1_000, mode)) + }) + } +} + +func TestInactiveStakePreservesExplicitCreditUpdates(t *testing.T) { + pcs := CalculatedStakePoints{NewCreditsObserved: 2_000, Inactive: true} + mode := RewardCalculationMode{FullAlpenglow: true} + require.False(t, shouldForceCreditsOnly(pcs, 1, 103, 115, 1_000, mode)) + require.True(t, shouldForceCreditsOnly(pcs, 0, 103, 115, 1_000, mode), "disabled inflation") + require.True(t, shouldForceCreditsOnly(pcs, 1, 115, 115, 1_000, mode), "activation epoch") + pcs.ForceCreditsUpdateWithSkippedReward = true + require.True(t, shouldForceCreditsOnly(pcs, 1, 103, 115, 3_000, mode), "vote credit rewind") + + // Tower does not inherit Alpenglow's automatic skipped-reward advance. + pcs.ForceCreditsUpdateWithSkippedReward = false + pcs.Inactive = false + require.False(t, shouldForceCreditsOnly(pcs, 1, 103, 115, 1_000, RewardCalculationMode{})) +} + func TestInflationRewardsUseHistoricalSlotTimeTransitions(t *testing.T) { schedule := &sealevel.SysvarEpochSchedule{ SlotsPerEpoch: 54_000, diff --git a/pkg/rewards/inactive_stakes_test.go b/pkg/rewards/inactive_stakes_test.go new file mode 100644 index 000000000..c1edc7c95 --- /dev/null +++ b/pkg/rewards/inactive_stakes_test.go @@ -0,0 +1,118 @@ +package rewards + +import ( + "crypto/sha256" + "encoding/json" + "fmt" + "os" + "path/filepath" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/Overclock-Validator/mithril/pkg/bankhash" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/lthash" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + bin "github.com/gagliardetto/binary" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// This reduced incident fixture holds all unrelated slot effects constant. +// The production reward calculator, spool distributor and bank hasher must +// leave the 33 fully cooled stakes unchanged. See testdata/epoch116/README.md. +func TestEpoch116InactiveStakeBankHash(t *testing.T) { + var fixture struct { + ParentBankhash string `json:"parent_bankhash"` + Blockhash string `json:"blockhash"` + ExpectedBankhash string `json:"expected_bankhash"` + OriginalBankhash string `json:"original_bankhash"` + BaseLtHash []byte `json:"base_accounts_lt_hash"` + StakeHistory []byte `json:"stake_history"` + Stakes []struct { + Account *accounts.Account `json:"account"` + VoteEpochCredits sealevel.EpochCredits `json:"vote_epoch_credits"` + OriginalUpdatedDataSHA256 string `json:"original_updated_data_sha256"` + } `json:"stakes"` + } + raw, err := os.ReadFile("testdata/epoch116/inactive-stakes.json") + require.NoError(t, err) + require.NoError(t, json.Unmarshal(raw, &fixture)) + require.Len(t, fixture.Stakes, 33) + + dir := t.TempDir() + require.NoError(t, os.MkdirAll(filepath.Join(dir, "accounts"), 0o755)) + require.NoError(t, os.WriteFile(filepath.Join(dir, "largest_file_id"), make([]byte, 8), 0o644)) + db, err := accountsdb.OpenDb(dir) + require.NoError(t, err) + db.InitCaches() + t.Cleanup(db.CloseDb) + global.ClearPendingStakePubkeys() + t.Cleanup(global.ClearPendingStakePubkeys) + + parents := accounts.NewMemAccounts() + var storedAccounts, erroneousUpdates []*accounts.Account + votes := make(map[solana.PublicKey]*sealevel.VoteStateVersions) + for _, row := range fixture.Stakes { + acct := row.Account + stake, err := sealevel.UnmarshalStakeState(acct.Data) + require.NoError(t, err) + require.Equal(t, uint64(114), stake.Stake.Stake.Delegation.DeactivationEpoch) + require.Equal(t, uint64(115), row.VoteEpochCredits.Epoch) + voteKey := stake.Stake.Stake.Delegation.VoterPubkey + votes[voteKey] = &sealevel.VoteStateVersions{ + Type: sealevel.VoteStateVersionV4, + V4: sealevel.VoteState4{EpochCredits: []sealevel.EpochCredits{row.VoteEpochCredits}}, + } + require.NoError(t, parents.SetAccountWithoutLock(acct.Key, acct)) + storedAccounts = append(storedAccounts, acct) + global.EnqueuePendingStakePubkey(6264000, acct.Key) + + // Independently reconstruct the logged erroneous write, and verify its + // byte hash before using it to establish the original bad bank hash. + bad := acct.Clone() + stake.Stake.Stake.CreditsObserved = row.VoteEpochCredits.Credits + require.NoError(t, sealevel.MarshalStakeStakeInto(stake, bad.Data)) + require.Equal(t, row.OriginalUpdatedDataSHA256, fmt.Sprintf("%x", sha256.Sum256(bad.Data))) + erroneousUpdates = append(erroneousUpdates, bad) + } + stored := make(chan struct{}) + require.NoError(t, db.StoreAccounts(storedAccounts, 6264000, func() { close(stored) })) + <-stored + count, err := global.FlushPendingStakePubkeysThrough(dir, 6264000) + require.NoError(t, err) + require.Equal(t, len(fixture.Stakes), count) + db.RootedDurable = true + + f := &features.Features{} + f.EnableFeature(features.AccountsLtHash, 0) + f.EnableFeature(features.RemoveAccountsDeltaHash, 0) + calculateHash := func(updates []*accounts.Account) string { + ctx := &sealevel.SlotCtx{ + Features: f, ParentAccts: parents, + AcctsLtHash: new(lthash.LtHash).InitWithHash(fixture.BaseLtHash), + } + return solana.HashFromBytes(bankhash.CalculateBankHash(ctx, nil, updates, + solana.MustHashFromBase58(fixture.ParentBankhash), 0, + solana.MustHashFromBase58(fixture.Blockhash))).String() + } + require.Equal(t, fixture.OriginalBankhash, calculateHash(erroneousUpdates)) + + var history sealevel.SysvarStakeHistory + require.NoError(t, history.UnmarshalWithDecoder(bin.NewBinDecoder(fixture.StakeHistory))) + newRateEpoch := uint64(0) + result, err := CalculateRewardsStreaming(db, 6264000, &history, &newRateEpoch, + votes, PointValue{Rewards: 12922370184029}, 115, [32]byte{}, + &sealevel.SlotCtx{Features: f}, f, RewardCalculationMode{FullAlpenglow: true}) + require.NoError(t, err) + updated, _, distributed, burned := DistributeStakingRewardsFromSpool( + db, result.SpoolDir, result.SpoolSlot, 0, 6264001, nil) + require.Zero(t, distributed) + require.Zero(t, burned) + require.Equal(t, fixture.ExpectedBankhash, calculateHash(updated)) + require.Zero(t, result.NumStakeRewards) + require.Equal(t, uint64(1), result.NumPartitions) + require.Empty(t, updated) +} diff --git a/pkg/rewards/rewards.go b/pkg/rewards/rewards.go index ce0e824a0..bfff0ebfc 100644 --- a/pkg/rewards/rewards.go +++ b/pkg/rewards/rewards.go @@ -43,6 +43,9 @@ type CalculatedStakePoints struct { Points wide.Uint128 NewCreditsObserved uint64 ForceCreditsUpdateWithSkippedReward bool + // Inactive is set by Alpenglow points calculation when the delegation + // has neither effective nor activating stake in the rewarded epoch. + Inactive bool } const legacyInflationSlotsPerYear = 78_892_314.984 @@ -697,11 +700,14 @@ func calculateStakePointsAndCredits( } newObserved = max(newObserved, latest.Credits) - effectiveStake := delegation.StakeActivatingAndDeactivating( + status := delegation.StakeActivatingAndDeactivating( rewardedEpoch, stakeHistory, newRateActivationEpoch, - ).Effective - if earnedCredits == 0 || effectiveStake == 0 { - return CalculatedStakePoints{NewCreditsObserved: newObserved} + ) + if earnedCredits == 0 || status.Effective == 0 { + return CalculatedStakePoints{ + NewCreditsObserved: newObserved, + Inactive: status.Effective == 0 && status.Activating == 0, + } } totalStake := mode.RewardEpochDelegatedStakes[delegation.VoterPubkey] if totalStake == 0 { @@ -711,7 +717,7 @@ func calculateStakePointsAndCredits( } } points := wide.Uint128FromUint64(earnedCredits). - Mul(wide.Uint128FromUint64(effectiveStake)). + Mul(wide.Uint128FromUint64(status.Effective)). Div(wide.Uint128FromUint64(totalStake)) return CalculatedStakePoints{Points: points, NewCreditsObserved: newObserved} } @@ -789,10 +795,14 @@ func shouldForceCreditsOnly( pointValueRewards, activationEpoch, rewardedEpoch, creditsObserved uint64, mode RewardCalculationMode, ) bool { + // Agave's skipped-reward credit advance applies only to effective or + // activating Alpenglow stakes. Fully cooled stakes must retain their + // account bytes, even though their vote account has earned new credits. + // The explicit forced-update cases still take precedence. return pcs.ForceCreditsUpdateWithSkippedReward || pointValueRewards == 0 || activationEpoch == rewardedEpoch || - (mode.FullAlpenglow && pcs.Points.Eq(wide.Uint128{}) && pcs.NewCreditsObserved != creditsObserved) + (mode.FullAlpenglow && !pcs.Inactive && pcs.Points.Eq(wide.Uint128{}) && pcs.NewCreditsObserved != creditsObserved) } // CalculateRewardsStreaming performs a streaming calculation of stake rewards. diff --git a/pkg/rewards/testdata/epoch116/README.md b/pkg/rewards/testdata/epoch116/README.md new file mode 100644 index 000000000..41cd2cb6a --- /dev/null +++ b/pkg/rewards/testdata/epoch116/README.md @@ -0,0 +1,49 @@ +# Epoch 115 → 116 inactive-stake regression + +`inactive-stakes.json` is a reduced fixture from the Alpenglow failure at slot +6,264,001 on 2026-09-21, running Mithril `320ce8da`. It contains the 33 fully +cooled stake accounts that Mithril incorrectly rewrote, their vote accounts' +epoch-115 credits, and the saved StakeHistory sysvar. All 33 delegated stakes +had deactivation epoch 114 and zero effective/activating stake in epoch 115. + +The source was the supplied `/mnt/mithril-accounts` checkpoint at slot 6,264,000 +and `footer-bankhash-mismatch-slot-6264001.json` in the supplied logs. AccountsDB +was opened read-only. Account bytes are public chain data; no keys or validator +identity files are included. + +To isolate the defect, `base_accounts_lt_hash` includes all correct slot effects +and the unchanged 33 accounts. The other 792 modified accounts were reconstructed +from the parent checkpoint and deterministic slot updates; all 825 reconstructed +accounts matched the diagnostic's individual SHA-256 data hashes. Combining +their deltas with the saved parent LtHash reproduced both the diagnostic LtHash +checksum and the original bad bank hash. Undoing only the 33 credit-only writes +then reproduced the exact expected footer hash: + +| State | Bank hash | +| --- | --- | +| Original 33 erroneous writes | `CCY2QcFoGGAodDaB9RCAW2RUbbZm2xMo9dJw8tKHdDcG` | +| Preserve the 33 inactive accounts | `BBXWdsTHc3EdN8zXbcZZ8vwqtYjzggD3gp8CqkZuBvBm` | + +`TestEpoch116InactiveStakeBankHash` checks each reconstructed erroneous account +against its recorded data hash, checks the bad bank hash, and then runs the +production streaming calculator, spool distributor and bank hasher. Before the +fix, that path emitted 33 zero-lamport writes and produced the bad hash. After +the fix it emits no writes for these stakes and produces the expected hash. +Other slot effects are held constant; this is not a full signed-shred replay or +a replay of later slots. + +Reference behavior: + +- [Agave ab655329, inflation_rewards/mod.rs:254–270](https://github.com/anza-xyz/agave/blob/ab6553293094e59dee7d3e7c928c7fa1023d0684/runtime/src/inflation_rewards/mod.rs#L254-L270) + restricts skipped-reward credit advancement to effective or activating stake, + preserving the explicit forced-update cases. +- [Firedancer 57d39904, fd_rewards.c:649–678](https://github.com/firedancer-io/firedancer/blob/57d39904e3886731b96b9174ff8763ee7c36e3ad/src/flamenco/rewards/fd_rewards.c#L649-L678) + rejects an inactive credit-only update. Its + [points calculation at lines 908–917](https://github.com/firedancer-io/firedancer/blob/57d39904e3886731b96b9174ff8763ee7c36e3ad/src/flamenco/rewards/fd_rewards.c#L908-L917) + retains the inactive flag from the existing activation-status calculation. + +Run with: + +```sh +go test ./pkg/rewards -run TestEpoch116InactiveStakeBankHash -count=1 -v +``` diff --git a/pkg/rewards/testdata/epoch116/inactive-stakes.json b/pkg/rewards/testdata/epoch116/inactive-stakes.json new file mode 100644 index 000000000..e6dec5b48 --- /dev/null +++ b/pkg/rewards/testdata/epoch116/inactive-stakes.json @@ -0,0 +1,1693 @@ +{ + "parent_bankhash": "B2Yi5ecGCiZRpvjdrLeQj2SdSncKgi6zEmVJxpMfzfyB", + "blockhash": "GBQcgk9JL3YFaK1GvpKLGpCbpmJ3My4oYWztnAJ3wirY", + "expected_bankhash": "BBXWdsTHc3EdN8zXbcZZ8vwqtYjzggD3gp8CqkZuBvBm", + "original_bankhash": "CCY2QcFoGGAodDaB9RCAW2RUbbZm2xMo9dJw8tKHdDcG", + "base_accounts_lt_hash": "Mnn6NyRVWRvlXK1fJ68LNsMwgwvv5fH/brKehzoh9D+CaKo7hT/Wa1o3UkLLI7xs14mAj7EgyHCauhEVsMQGISCDkpWwejJy5tH/1Ov8yb+FM+1ZkSIO1vL8vsxDGipgAGfNXO+nGnh/ldg6Jbv66RgGdXJtR4b2IMUD970fj9z7229YpWP7PIwy70TEJ5VQuS16bXRwJlA8cRSSn7RMx9/yW2G4WhS+XAGi13N1uPsIHUkm6DePr4WPRsUi4UjMP3F8fpYUCAyVi8NMeFBPPGHZa/ZkR2C3NTk1Fs/jpI3UTw7MI61RnDFZs2vd4YYx6gr2g2jRKBX2/3NgSEK0112FCnB6T9V+LM6b+flCUlL5OddAd5MtN+4gDN2DlHAX20jb8VTWqwvb602v877ha/L+YK09+D+Gr9+iNxx8CSCAycb9moun0LGVSg6TgenXfhwKQbYlLtbIB7bD6YCrLBeP4A9rifOBIobmidjeYBjF7kicq8vMrH3j+flLSsn4fHXsssP9edqitP69ycRm7nLrYEFTR5+tl985wzrWi9DtOnImREgGo1oHBqGxXe6BksLdnCSei1i9SOX3oEaxKWf8bcf28ocL3pf5JKw3cBNALWTdjda/w/CMkg4N6MB6pUurImpm1M+Sg9O2Qyvv6v24RgG/TBdqMTmt3b46einSa+KZO72BQ+c4KkihabsRBpJNYpLQAiSH1Xm6FWoSuu63pRnGRyYokF3JC7K7M5wYpqE2EYJUUVM/SH7uS/JhOmWZXtGSpBcq0kQjHfIIjsJhrD2om9NJgVxEf4DhwK3ZTHUdTE5SY5aOwB/HVNlaTQd2Ab1ZsJncKAGIz1LcJpLB1uI3CtjtTwKUYrYTukHpWl2OQySqy0gmEzNPqmDUbhMUuEHcvJBUw30wLusR/BmlR9RErlxtigecyID+K9O9sTssRnAfuCY6Hm9SQP0rd6VRbIDqcfvSp4nOdYIZhx84IcN3SkN2Q9ipCc/OBCf0b2vIUu/7YS7VPv0gUQCKgg+kHJzZzUD5vHsOTw2623uxug2Y3tkiaVgrh861s2lJTjsSz351rMZQc0nzjis+GAoOuzo34zpGX3duLHYzxxdiRcX0jSvM4C9CVX47dQsYnltzVgOfXbeLYxA19SBjrhErooboDLe5OfeFeHGMIUO01sQ7dXXTzLZt03Opb8UjTbAtV3zCN4l08vObalYPFm4zEuTGYVnWIrJp4d2sEg/p7ejmWQNif/36zSLANp6ILgJqbVp4uATjuPPQD/nEcqlTspMy3f9yDK2Du3DjOcRNR6w17AtQvKsyH5dNo3w9JWG0a9HeBciz5CwdqYZ8cTi9jouD9Yj8roxAH9l63gNNJ+RgO62/17EBw9fS3pbNDfyJbZfvbnFq1KpIHe0LqK0K7hL7zTpcu/KgDKJLLZZe1YeDUm8jlVlhpbuDCBr3oj5N2p37581r5thU0F4f8//vqHm8G7i2UWt7sXtxJJ5JhUDV7RLYLfmFRc7ZXBJ7Z9qAxTo4WIWWfb3QsxSgZK2VfBFerxHzE8QlDCukgAHnjRxx/sVAHU5kqTI+jHpXsDogLmCTvaTuIW7x2WAmdEazWRi//5J9COsBt3wOwQwwBq7nRfj8w7Zo/M5xwbdpTv/JsvLQKmc51uaYVBSHZy8q54g7jo+j75jiA6TCh4KzMejWbH+c9Ew44ATW5mLiPFCJxmkZLI9Dl5XXxTtspSkHlTinziKS8+FCFUD8iCdnXKkym5gD1oO0imsGvBRs1qqnwbIaOv7esnnh4n8V3ML5La3du54etTnaGlAI35lGv2AEc2otPbHDvffI3PsMuEi9YkktqO7rJmFieUIU/uNiMevudlQqqrWM499kCOI7C6drxi8KfdcCd9b88nGSL6MlKu7F0cwttJWCb2uDTWEA2vYbxpUtDxZ8xYNeCdKDHJbNGk9viaLneQ0BzE3GQLnZPEYAuVkAbh+mWN0o/kmL8DQF4wpa1xBFUBBD6zhhM5zL6yAF5zT5k+bGkrIGS6+D0OteyVZnB+2y/ttjRknqhc+5CQVr90dEBsk+ivEsz7Ta/vZ4ibYJOgHq6z4OfagQWaWCiYBaQoTfKhLpekY3HrIMhQSsXuH8UjiRULhj0mqLnX7UarLLCTAvqEsvCgQxH1vzNbeomzI3HlJtWTrQz/vv8DOPXzJE/aYuxbyTAhq6uZH6TkH4qmv9lOh5dLbtQVZq1uf0upbJgiamwRyf+PGKqBe+rcYz3Mp/2aJk+jAYx4WGVdd66COzTvV///vH+M+/E7Yj4ueEnqsQVuLMtjAW+xTu9UiA/eb/mMVBpzeRaOUELDeYHwiKR8HiwLH73UO2AGuFsCAhqQPsYP/SSOaKvJinVKe7uclMgEgSAknr2UTxVjoPay18vynGYJOYDBrxFaxaDQbCglcjxfJW016E9SGRbxxlZZOqpr9qZtO20MgGMsgotBKniJ87ssxlJDJ+Kx7w2MWcIN5dv7kFj9KVIiBJ206fg6fQUHB+/4g0x7rBedhfmDtWizaNyqNcpWHfRMhoIPI+nOij3u8ASv/Z5HgqvFDeqn3MBix5Z/4+vRpS24pN83kvdbaGnyLSZ6jNuq2wfdN2PYYlnUczLvrwD7QjUkf+mf3PVhZxM+56+lnYG52tToqXIqAtrq4fkJq73lCOLwnd+8cCkdbQtx3qQF+NiR1eMiewNwz6cVTKoMdNqeYsGmXRNMY=", + "stake_history": "dAAAAAAAAABzAAAAAAAAAJ5AeDrqWjwAaanDLwG0AQBXHmbgZhcAAHIAAAAAAAAAJXm6zkOLNwDYUEVSW28GAPX9Y+xHMAAAcQAAAAAAAACbOD5ij1s7AH8vYWThAAAANhpRt13RAwBwAAAAAAAAAEKuN0SOQEAAGg55mi/jAABjcyFX7dUGAG8AAAAAAAAA243TwqZJPgCgvqaXc7gCAGxJJAa+wQAAbgAAAAAAAADlcqREpWI+AF6z7ht6CQAA2yghg6ciAABtAAAAAAAAAOSauglaSj4A7jLIrkcZAABAJ1t/KwEAAGwAAAAAAAAA1bfp3PxJPgBswjk12BAAAMxhZcmqEAAAawAAAAAAAACgpYOCjzs+ADdbCt+GKwAAyzEg90YdAABqAAAAAAAAAHdjrA3dQj4A26mxDlISAAB8hECCzhkAAGkAAAAAAAAAAEdfVIHpPQBW2hsNHo8AAC+XfvrrNQAAaAAAAAAAAABo9vZqfuU9AA58Eg7PBAAABIpq+vsAAABnAAAAAAAAAJFweNQULT4AqV72RLRIAAD2LXPQeJAAAGYAAAAAAAAA9qEC3Yo/PgCF7iNpyQ4AAAZZwTZvIQAAZQAAAAAAAAAuQbhfCTk+ANbBFd9uDAIAstBbUhgGAgBkAAAAAAAAADB84TH1QT4Agz3YD9wMAAAUQT/28xUAAGMAAAAAAAAAgc99eYkCPgDGnMraNEECAIGF5S/2AQIAYgAAAAAAAABNb2WWlro9AGWphlSdhQAAyYxLbNc9AABhAAAAAAAAAFmW5xtaxj0AF1Ks46MDAACV5g+0lg8AAGAAAAAAAAAAQgPK8m0kPgD7oWlICjIAALtPvWhHkAAAXwAAAAAAAADvAGYj8wo+AHtuvjg5PwIA4Gqgee8lAgBeAAAAAAAAAPRPsb1kNj0A6J0nTzorAwC0i2yT2VYCAF0AAAAAAAAAIiDIxHosPgBZmiIiYSYAACkVK22lHAEAXAAAAAAAAAAJvmQRggs+AAIGnLbqIAAAWMutxRwAAABbAAAAAAAAAL/QM8VAMD4Aw7etwKcAAADauqqckSUAAFoAAAAAAAAA65RRlJUVPgDMSCTC02cAAIfN5q1YTQAAWQAAAAAAAACP/BZcKsU9ACAmYvigdAAAiDfz8mMkAABYAAAAAAAAAEK6x8WZyT0ArEiebikKAACSNlCKxQ4AAFcAAAAAAAAACTBqaG7SPQCoR3XOSAsAAF71j7lJFAAAVgAAAAAAAAAjWmA61589AMtdU9w4VwAABJfvbs4kAABVAAAAAAAAAG/JhOUTaT0AUMvSkd44AAB1Iv0rRAIAAFQAAAAAAAAAfK7kidvtPQBNKEsZw0gAAC4+dp63zQAAUwAAAAAAAAB8tlqi8ts9APvZJnnSHwAA/WCY9BIOAABSAAAAAAAAAMAYELB0AD4A+dGs4HEaAADOy8rLIT8AAFEAAAAAAAAA0TQODTb8PQD+ZIrMvQ0AAKLsbq2qCQAAUAAAAAAAAACb6i9EQwM+AEicn0T6PAAAyAM4YTVEAABPAAAAAAAAAJeDquB9Qj4A29/hUTE6AADzZwr3lnkAAE4AAAAAAAAA3YhSTtglPgA1M80MFSYAAGDVIn6dCQAATQAAAAAAAADavbp48hw+ADUoxH3YCAAAmmSrqRsAAABMAAAAAAAAAM9KZ53SHT4AFPzaJG8EAABBh/RQegUAAEsAAAAAAAAA+HspIqN4PQAJ3QG6ZLgAAIXaa4phEwAASgAAAAAAAAC0Hzu/esY6AHt5mUByuQIAmiDhzXUHAABJAAAAAAAAAEMqtGP2cD0A84gQpRYFAABAZYJ8w68CAEgAAAAAAAAADk3OwuxvPQDAS3n7GBkAAE+VB8A4GAAARwAAAAAAAADRsZ0Vx1U9AHLLb4SiHgAAZfdTUKYEAABGAAAAAAAAAKNb7H87Qj0AZZCx2/oiAACT9t0tmQ8AAEUAAAAAAAAAQvmdRxdOPQDVa544vgcAAJYPouDEEwAARAAAAAAAAADx6iXLmDU8APdU+ynaJwEA6/70KIYPAABDAAAAAAAAANZyds2GODwAFLJEEZYGAACa0y6VrgkAAEIAAAAAAAAAe6+cLig2PABhXBlazAsAAC0teOaYCQAAQQAAAAAAAACxff6Ze1E8ACrBuEaqiwAAJkWtgCmnAABAAAAAAAAAAPLJlmbtbzwATFPM6gYKAABz5YOHoygAAD8AAAAAAAAAlXqLG0GDPADGPdE0LRAAAJvWmUOwIwAAPgAAAAAAAAAGKrZkAoU8AFZorVhRMQAALFQFTUEzAAA9AAAAAAAAAAoUqYH8sTsA4IDLEML4AAC9/5aE6iUAADwAAAAAAAAAuzL+y6ZrPAAB8TW7cSgAACoQ4FtL4gAAOwAAAAAAAAAn1bWd5RM8ANwhM7NEiQAA9kqDwa4xAAA6AAAAAAAAAFxafIPkIDwAK2eWKPQ2AADVJYEPHkQAADkAAAAAAAAAViHBYFraOwD4j7g0FpAAADQ7TdK4SQAAOAAAAAAAAAC0B1p+6RA8ALCVqnGzOQAAclnWd2pwAAA3AAAAAAAAABJqVZSu8TsAxZ4RSV55AADnTpfITVoAADYAAAAAAAAATOvQCGvlOwAInOaCLw4AAGlNI3AVAgAANQAAAAAAAADiOkL7a3I7AP3nWviafwIAPurnf8cMAgA0AAAAAAAAAISnBw5hVDsASNjGSPMhAABXGeKPGwQAADMAAAAAAAAA75Bzrs90OwAB3HKPxhQDADDloUJeNQMAMgAAAAAAAADdNGOwmVQ6AKrWSFPxbgEAhqz8EelOAAAxAAAAAAAAAKVlV2M5bTsAt7J6MWEZAABWgGutNTIBADAAAAAAAAAATdF/eGFmOwCMYa1ndx8CAFnG6qTIGAIALwAAAAAAAADaHqDgbWs6APgZHPi53gEAfGTrf/rjAAAuAAAAAAAAAHVDwpv5ozoAnUFqwhSoAAAwz2LP1eAAAC0AAAAAAAAAaE1WTy5QOgC7Lb6VWeQAANrlRLzBkAAALAAAAAAAAABMwuR/ciM6AEHpLe8wUwAAcI0k8acmAAArAAAAAAAAAGUV3bRCGDoAkDAWSOEkAACeekS34RkAACoAAAAAAAAArx9e+fYeOgB8ySqUlQgAAAs3i/h6DwAAKQAAAAAAAACNHcp9cl06AJaDFRAWbgAATpEP8MasAAAoAAAAAAAAAJSotzIIpzoAFwSV826JAAAy92cXMtMAACcAAAAAAAAAPFUSdQ65OgDoJJ6NhCcAADCwfvHBOQAAJgAAAAAAAAAA6p5t5546AMm2R+E9JAAAo0xN31AKAAAlAAAAAAAAAO6h+UfxQToAnx24iv5qAAAddIZ3PQ4AACQAAAAAAAAAKa2tRNmFOgACcRzbgS0AANYPtIOhcQAAIwAAAAAAAAAGOyVz3zI6ALqiF0x0+QAAi9QI3qumAAAiAAAAAAAAAOcmAEoM+jgA4sBkWs5BAQCzLciCLAkAACEAAAAAAAAAdn8sBEvPNABKZYiSUOEFAKq488AelgAAIAAAAAAAAAD0lsvPNDw1AL9v6SJ28wAAlbOWU4hgAQAfAAAAAAAAAGk2uzPpvTUADkS44IFUBAD3aJUNxGQFAB4AAAAAAAAAG4YELojOMwBT2ECjQoQEAAUoip3hlAIAHQAAAAAAAADYscqSXXQzAFufDzqhIgEAGMvVnnbIAAAcAAAAAAAAALtrLOOo4TIAHUaer7SSAAAAAAAAAAAAABsAAAAAAAAAPlfpgiWuLgD+x4A4R4oEAAAAAAAAAAAAGgAAAAAAAABhNJyrrrQrAPe6b3aatQcA4T05ZoT1AAAZAAAAAAAAAOgqRqBFESoAOA1GOSyqCgD7mGj70iUCABgAAAAAAAAAQUvKQ90zKQBbee1BlPgEAIO7vG7m1wIAFwAAAAAAAAB0/UZWrDMpAFfZoR9TcQgAbNhDyqKIBgAWAAAAAAAAAGTa4bqZwSoArSc7BflKAgATQNWp51cJABUAAAAAAAAAuJa1PExVKgAIutFU5XQAADZwSSvTCAAAFAAAAAAAAABdiCSRT04qAEvh4+XMBwAA6Ma1tAYBAAATAAAAAAAAALBNfKM56ykAxKmk5NtrAAAgpzgl+ggAABIAAAAAAAAAcrdEoMDgKQCuPiLWZmYAANwwSk8mXAAAEQAAAAAAAAAUXTuSI9EnAILkep/EswIA4YaSbF2kAAAQAAAAAAAAAEhOd7PGYiUAJd7/3yVuAgAAAAAAAAAAAA8AAAAAAAAAVP+zxFlMIgDlgfGO3iIFAAAAAAAAAAAADgAAAAAAAABT2ixTNXcfALK/SZfhyAcAAAAAAAAAAAANAAAAAAAAAESpbbv03RwARWz5UURhCgAAAAAAAAAAAAwAAAAAAAAAQFrTd6Z7GgDMXyiGppQMAAAAAAAAAAAACwAAAAAAAABcnB0IxEsYAH3pX8hj/w0AAAAAAAAAAAAKAAAAAAAAAJCaKkMgShYA38jJYAvPDwAAAAAAAAAAAAkAAAAAAAAAF58Ll+hyFACFLkwOVlARAAAAAAAAAAAACAAAAAAAAAAxtAZmmsISAEKE8AX0/hIAAAAAAAAAAAAHAAAAAAAAAHXDQI4BNhEAACv2ahmEFAAAAAAAAAAAAAYAAAAAAAAACt0HxSrKDwDjq1Ie9r8VAAAAAAAAAAAABQAAAAAAAACxrhYtYXwOAN0of1y9rhYAAAAAAAAAAAAEAAAAAAAAAC2YPXwpSg0Ap97CpRE4FwAAAAAAAAAAAAMAAAAAAAAANf2HdjwxDAD7OmBvB/AXAAAAAAAAAAAAAgAAAAAAAADgL1gZgy8LACVZ0BHMARgAAAAAAAAAAAABAAAAAAAAADurkkgTQwoASjP+BeaxFQAAAAAAAAAAAAAAAAAAAAAAgGxxLSlqCQDgMpi3KeAVAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA", + "stakes": [ + { + "account": { + "Slot": 6210151, + "Key": "3mZwezEBuKcCs42NfAEZfdCbCMyPaV6wnus263mgrhjP", + "Lamports": 59002277492, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAI9JSTu3nHzfydAzLjeA87Aht6Ost2Mol90Y42gdbSqB9HisvA0AAABnAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACZzNzwgAgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "7e91c9547ff107e827f22f226df525ee83be7e9eb22bc8afd812c92c235c69ba", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9430889728015, + "PrevCredits": 9349889838233 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "453ETE3U7Usc4ss87sszs3EvTR5ZYQQWcZHgs9XiAZdY", + "Lamports": 3306766663971, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAADSIHvAB+hqCeyUMqrtA7RF7gf0pF5+WOdx24l28NcLZoyuE6gEDAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAALJPnDAREAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "88983f1b8c3153b155fc54b8ffcdbe468b3356ce007cdab77d95cab1d3c50dc3", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 18772882129308, + "PrevCredits": 18699280524299 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "462wmoxLUzHfxcURxwkMj5f7cmeVXeSeaFU5uXd9jrQE", + "Lamports": 30092278839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAB7glTPNCA5Xp0adNqBsSaCIXSwoB3DhCXGyxm9mwIQut+aAAQcAAAA0AAAAAAAAAHIAAAAAAAAAAAAAAAAAAABJX296lgQAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a0c0c400478bf441c374e56141b0122846da672b9b1a5d01157374dadfa83649", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5082302209818, + "PrevCredits": 5044345724745 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "4aNZYt3giNatv4J58sYHwRNs8L6ZQpLqG9GZiyqrXUqS", + "Lamports": 2329209663118, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAK8Fa5ghODdNV981iyfO/aADxuCTmU56Dtg15Wf9Xk+LDhmUTx4CAAABAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAeHbzX0wkAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "78674eaaabfbe62cf2e558bd33e01dd53a33a7a4024e52eba673bb1acf73ff83", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 10909280173169, + "PrevCredits": 10805462179102 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "4u4XDhXRtXA9QqP8uC9vpoyw4KLaeWHm7g2gP4aLDGK1", + "Lamports": 59002277492, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAABtRAg7KdM61d5wq/bIMit41UurMHM2/gwzsJrDkSJWr9HisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAC8dSEI5AgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "f49eb225b3f858e7609ffecf5e9d2734cb28fe754ca697800e0a44a144ab0f3d", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9855957906862, + "PrevCredits": 9775481976252 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "4zrGtmZj2s7akr4FyiJ7E3fQY4n59cZoVEmL192GttRq", + "Lamports": 3956852941834, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAApBQ/86W982eATp1soYW/TvCCjcBXX1dK637et+U7Qaio6tRpkDAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABz+BFZ1gkAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "6a0c148dbf978bfec9e6e5e3227ad57e6089c056b2d23730b85be7bdcc2a3384", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 10900522634265, + "PrevCredits": 10816222001267 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "5ZEUi72iZM7ozwisZV2jezjgR3BLBQToz8ZxpdnqYaue", + "Lamports": 10521999279304, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAEvFm+VinHP4nTd2zkbv4cVplRt3TlVeTEidtXhPv7vISK/k15EJAABiAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABDEW5ATQcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a0c42ad17b70bd37283facbe1655a9401660d0ba96e7c57fad1ba452c46c77c8", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8111781721836, + "PrevCredits": 8028374831427 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "6dNDcZRYeX7zoGirAbYxjJVCoHg14KLcy1ndb5a6hLtp", + "Lamports": 92402274844, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAALJZyCh5aoPz7tWdy6VqoilN4lxeoCa10yfJpJBoYSa8nPx3gxUAAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACEXlHvNAgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "4914fbc352aae3350f8dd39a94733fc3bd927e1ccd2ed471ac81318d759594f9", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9114167602897, + "PrevCredits": 9023446408836 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "7Xwjq1Pu1ibBufp4mdu51yrRzUoadak98sdj6BQhGmBH", + "Lamports": 59002277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAMb+8swy4nz1EbP63bJ/XUX49DQYo/CiTC3dFaYhgfYs9XisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACE/LmcrgcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "01df0af720551818974311706b7a5ebb9c5b333eb90e877e0f227f08fd4cf9e3", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8523194289275, + "PrevCredits": 8446535138436 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "7kGbq8H94roSeiCvasCE8kWq2Kx1dCTZQTUnkiPqibTg", + "Lamports": 59002277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAaFSKZ/FpeRshn8N9+v+AkQycTuIPwqrs5Yf2mgYnGv9XisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAF9xJssgcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "f04a2844299ed1e7c980b160a89a508f61a135ee5d333ddbb948aabb00b80a6d", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8544891256299, + "PrevCredits": 8462898755333 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "7uyNRdfhmk6SwJPfM172LF4ajEUKUkP9RC2okeB3RZsC", + "Lamports": 48931548636, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAABLFnpY8I58W61QVYzmx6wfM3seg+Q0PxwXVhMb9YvpaXFhpZAsAAAAzAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABZwrzIpQYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "d2a561a19b6f7cd09f21985876978ae5ab1e11c7cc9262c5a17caf9af6e0c92b", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7371534095782, + "PrevCredits": 7309107184217 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "8QYDcU8pvmMKvovzUiqWq4H7ZFiWpbet4HwUnDy5XTYK", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAFgYhGqxkhQFABeF03ECpK9RUUF1hdavhflLAu/MvP/rAMqaOwAAAABtAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAs5ZJ1sQUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "ff84dc2a8426d4deb78786afc87899cbb4c052b6c20346fac76f7fcd865026c5", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 6318982736717, + "PrevCredits": 6259739911468 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "8n3LHx61MeuZTF28hkHvsiEtKSEXLUQDYCMHS8sCQ264", + "Lamports": 3058382969924, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAM9GQ73y9miIEoMD1/GtvVm6Z6RC6u1sNUKM6BM6jWJ0xMaxFcgCAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABQGTGKHhEAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "20b17fb68d98d290dfa4742237ac719219e23147a8f7cdfb27ac1be790f9e8d8", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 18900384617754, + "PrevCredits": 18822865164624 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "98joChCUmuTBYZx8xHDJzJzbF37HPuP4EHqm3BwD8KSo", + "Lamports": 132002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAite2/mh1w1kW4RMXwAjDtR4kt3TcBCr/oy5ngX7hOQACjQux4AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACf+Oklb0QAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "7c11ae1d9490b3f0876c515a564337a0170c21fae8c3cb3f43c66e5dee2cd34f", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 76179624781196, + "PrevCredits": 75244168149151 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "9PAPBcFDq536ZzMWhNywZAYPZt5mF92VadrJc5ctoQqY", + "Lamports": 1894659468576, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAALo09fj+TmHMhldcHlC9iwfK4NZlNl+rkERD8XdoYYH1oFdeIrkBAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAC9udrpXwwAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "ba19d0de0f1d7ccded2e93dd4b9d57b8077397c396005d1918b4d6dd719196b7", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 13714985011803, + "PrevCredits": 13606084852157 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "ALrrFs8DAkNHjyce6LhwkotcBQp8U9dGKb4prUHZp5GF", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACZSY+h/WqtJAAWy9FwQ5zvCPyGbQUYwp5yepcxIn0qbAMqaOwAAAABwAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACXOQj8AgEAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a96af37ed2c6d36cfc77971b4432cb7c62f4e045b3883a5b7e2e6434fa21dec9", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 1147511243813, + "PrevCredits": 1112329959831 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "ARPzersuLPyg9ZJaxYuKc2eDKKg8t4qWXpTy9TzKjLJU", + "Lamports": 44842277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAF6PbC1s5mob1cKneVa1heDI7d/UjwFUwoZKG6DN1okV9QSscAoAAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAZRdBpGAcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "7c3b359b3e2c72b291d1703977f464395b1772f9197466913c9f87661d3ebac0", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7864995243760, + "PrevCredits": 7801435866393 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "AXC6s8QGstr1nnTJ5ranGEeSZ7rcHKVeQav4r4J8icJL", + "Lamports": 2951443346384, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAD2DXLDpqE+RtusMr3795aLUMJhXsVQnXUHgtyhqt1j0UJ6YL68CAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACQUqzjgQ8AAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "0e44daa0385ca084667f7e44f0fb96571b6b963f6bd046bf66c2c7fb31e232ce", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 17293215270489, + "PrevCredits": 17050544919184 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "BWiQJLo1TGhbNp3ionXVFPnSgfGimGU9FVHm1hdwLKbJ", + "Lamports": 30092278839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAADUkqa5BeE51KvSv12H//GyC8rI6ScrXABI9UawzUXiot+aAAQcAAAA0AAAAAAAAAHIAAAAAAAAAAAAAAAAAAADlkymHrAQAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "8cf243491cd54f9d1efc732287caf92bd392dfc1ad410c74c94e2d8f4e4deb18", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5181125525357, + "PrevCredits": 5139048535013 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "CbFTDpV9HjcnVQReCUwuZmLMp6PjMoWZBA7VTvLzszQ", + "Lamports": 5114239502629, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAANp1F9fR0omLpsfk9YL0X7w0EuA6D8E39bIn27KGqRgppfNKwKYEAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAADfLChErwgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "6bf6057cfa3c4d2a7d786bb5f325032604cca8eb30391468f2caf1b71406ce57", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9639820890790, + "PrevCredits": 9548855782623 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "D5Zx5bsdVDg2kBZkPvaF59ULdKd8LACtScaWE7NEAbtw", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAANbDQeBxoTnMt62Ygze0/L0wq3Tv5ynlZXdjhuu6lY/4AMqaOwAAAABuAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABO6xz/EAYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "2480dc54ffb62581eec0ca3bcb7201b2ba4260a97bdc186e2f2345241abade3c", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 6731352374437, + "PrevCredits": 6670069328718 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "DNCCPW9Bsv8SdcDMFDA7b2HT5jKxgUnBrA2gkKXwe5Bg", + "Lamports": 44252278839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAEZlguB/hNNN/i6p/xPZEHXCTr6qIxawtvTnx9A4WE77t1qBTQoAAAAWAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAJpOYcEAUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a02301fdf180b2f542124caf7da178cae4397acc833030b9dca28116d6e2b747", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5630002073869, + "PrevCredits": 5566762492937 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "E2TDLwJNmzvg6K7MhrkQ3HV4n4mp2zL49FqGwV4tFUfC", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAIZbuXtVwiaoNZBvLxeR/DEy46f1oQ/1DEqV1R+wxiWyAMqaOwAAAABvAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAA+JHwoWwgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "2ecdc0e446f6e0c112c947c79b6976cbdeea776ecb4fe62ece89b1f725e3ae48", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9263388828449, + "PrevCredits": 9187614270526 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "EA3MVKFbieGtDZDG4bmCiwnE2HWndMuWcjMjRNSrbMBC", + "Lamports": 3787600669028, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACGc29KCGn33qCWHYLj1Li6Q1LaT9ffcV8Q3PJ6AssCM5NN03nEDAAAGAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAChfXpwQQkAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "c0bf9bd502e0b675a186b4b855812ae1915d04110a951499d911e6174ed2b2b2", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 10252916692527, + "PrevCredits": 10176664599969 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "Eijmmf2xFvjStkcu6WQRj93x5iYP1uv5zuZP8PARdqd", + "Lamports": 3449661803946, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAC3bR7fFvVz26LQA9LQq1JGFKWXl7hg6BaC5sp0SpTflKvi6LyMDAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABF19K7FxEAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a4810c657cb2161f3c47f0452d78dc74d9d4597763d82c3dfd48b3d1dc43dfe9", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 18871239559140, + "PrevCredits": 18793633077061 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "FxhNhXDkoqMTcXqyHxYXW8BJAdpWH5qdyz3qgpiySrKn", + "Lamports": 71250450995, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAHeMPT3PZmkZZcARjam+dHkyZP326BpHWqy/ktZchIums8S4lhAAAAAgAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAXAHi5oAYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "ab580bd8bb513651006208d529d58dbe8e40d0306141e4b9bbe32c0559040af8", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7382306449058, + "PrevCredits": 7287376183319 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "GAyfc6a5E5uzXQgQ4KXHuJyunyGNtLGZgUUdWKEL4Df9", + "Lamports": 59002277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA0QDCxOaKoUheu2WKN7BFVixKbK4ogx2BYksnFTbTzL9XisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACYANgjswcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "d635c01a4fc8bc710b4ca6d9b0b0ba960b9ab3ad8ac0359ee8adbdd62ad30f3a", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8554183692312, + "PrevCredits": 8465981898904 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "H2dM1xYSeuZaujtByMa4jvPwUVYjfxzUL3K9iVLaCW9G", + "Lamports": 64469922880, + "Data": "AgAAAIDVIgAAAAAAzi6d3+ZZh+CVMQop8DVJNLt57mKRukRhiZHz0dnIWEDOLp3f5lmH4JUxCinwNUk0u3nuYpG6RGGJkfPR2chYQAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAiQo+w+o08xym3k1JBF+HYHIJZGTDaZaufK/simzs1bwB6SAg8AAAAgAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACe3jqUQyIAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "b39033f338419cea17108c3989a6331d8272749095fcf3ad856031cbab264110", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 38085420659301, + "PrevCredits": 37673645039262 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "Hh8u21hJnAvE9PMNBmNTsDUTseY6bYfwvbZQgj8R2vcU", + "Lamports": 44842277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAJOHGqQMr9AgJ5hIj044ilcd/38t2ijdTubnJyZ6iT+d9QSscAoAAABGAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABymDuk3wUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "0569f4f3dbc16ab5c284f6efaefd1e2b6b5b8db18ffa1b1fbd0da98468c54d78", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 6525335322632, + "PrevCredits": 6458091214962 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "HicPaN5ogXhE5jQ99WotjGUQLPJ11bdCqXjJhMKUq5N", + "Lamports": 1690908095830, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAJ1KF6URsOxn3mdoAgUdjof1ae904ReOis3chjrZx0h+1h/XsYkBAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAADt4ogGQgUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "47ad9ddb61ff1edee564852ac0e48ff45b47a377b0ca1a5f1e0aef74a27627b4", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5826444097536, + "PrevCredits": 5781135614701 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "c6Ev8GfADHdrGtvH1VqwH7JmN28PisZgGfUy64RE7np", + "Lamports": 3782196946839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAMDb0jorArBVkC7lcchWRZHVxqD+PrkqtMdnxFvaxkbfF5JenHADAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAA1TQUrswgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "b361cd2d770b1b3432d58f8098c654f300754fab58787d6033dae1e724618467", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9647673972566, + "PrevCredits": 9565613935925 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "eHboMWq5bmx1DxdKw79zpse75XLk2GEgeRnJG9nysJn", + "Lamports": 5125666350406, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAPznsTwPeDl/u5+jig9MvOBeS/RKUvjM0z9ITapNTvPoxs9iaakEAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAEpkAsaggAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "8e052b16e36ecb6e80919ada5456816e25e3b804fe9119319cff9861e7242e25", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9334067720448, + "PrevCredits": 9252101989892 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "v5pSSEZAuvG9ewGhdMNVcTVHeJPzFnGse3mX4FBEyLT", + "Lamports": 1082110856099, + "Data": "AgAAAIDVIgAAAAAAzi6d3+ZZh+CVMQop8DVJNLt57mKRukRhiZHz0dnIWEDOLp3f5lmH4JUxCinwNUk0u3nuYpG6RGGJkfPR2chYQAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAHeMPT3PZmkZZcARjam+dHkyZP326BpHWqy/ktZchIumI3ay8vsAAAAgAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAXAHi5oAYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "053f427dd888bbe45c6e0f772faf6eac1f9ede5822cc74477c757eaac75cb8b1", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7382306449058, + "PrevCredits": 7287376183319 + } + } + ] +} \ No newline at end of file diff --git a/pkg/sbpf/fastmem.go b/pkg/sbpf/fastmem.go new file mode 100644 index 000000000..1f15dab3b --- /dev/null +++ b/pkg/sbpf/fastmem.go @@ -0,0 +1,62 @@ +package sbpf + +import "unsafe" + +// memRegion describes one of the fixed 4 GiB virtual address windows +// (rodata, stack, heap, input) as a contiguous host buffer, so that the +// interpreter can translate the common case without a function call. +// +// rlen / wlen are the readable / writable byte lengths (wlen == 0 for +// read-only windows). gapShift/gapMask implement SBPF v0 stack frame gaps +// exactly like Agave's MemoryRegion::vm_gap_shift (gapShift = 63 and +// gapMask = 0 for windows without gaps, which makes the gap logic a no-op). +// A window that needs special handling (VASA input regions, ...) has +// rlen = wlen = 0 and falls back to translateInternal. +type memRegion struct { + base unsafe.Pointer // host address of window offset `start` + start uint64 // offset of the region within its 4 GiB window + rlen uint64 + wlen uint64 + gapShift uint64 + gapMask uint64 + dirty uint64 // 4 KiB page bitmap of fast-path writes (stack/heap only matter) +} + +// emptyRegion never matches any access. +var emptyRegion = memRegion{gapShift: 63} + +// A uint64 dirty bitmap can describe exactly 64 pages of 4 KiB. +const fastDirtyBytes = 64 * 4096 + +const numFastRegions = 6 // index 5 is a permanently empty catch-all + +// fastRead returns a host pointer for a size-byte read at vma, or nil if the +// access is not covered by the fast path (caller falls back to Read*). +func (ip *Interpreter) fastRead(vma uint64, size uint64) unsafe.Pointer { + reg := &ip.regions[min(vma>>32, numFastRegions-1)] + lo := vma & 0xffffffff + inGap := (lo >> reg.gapShift) & 1 + // Truncating to 32 bits makes lo < start wrap to a value >= 2^32-start, + // which is always > rlen (start+rlen < 2^32), so one compare suffices. + off := uint64(uint32((((lo & reg.gapMask) >> 1) | (lo &^ reg.gapMask)) - reg.start)) + if off+size > reg.rlen || inGap != 0 { + return nil + } + return unsafe.Add(reg.base, off) +} + +// fastWrite is the write counterpart of fastRead; it also records the dirty +// range so Finish only needs to zero what was touched. +func (ip *Interpreter) fastWrite(vma uint64, size uint64) unsafe.Pointer { + reg := &ip.regions[min(vma>>32, numFastRegions-1)] + lo := vma & 0xffffffff + inGap := (lo >> reg.gapShift) & 1 + off := uint64(uint32((((lo & reg.gapMask) >> 1) | (lo &^ reg.gapMask)) - reg.start)) + if off+size > reg.wlen || inGap != 0 { + return nil + } + // Mark the 4 KiB page (and, conservatively, the next one, since an access + // is at most 8 bytes and may straddle a page boundary) as dirty. + reg.dirty |= 3 << (off >> 12) + return unsafe.Add(reg.base, off) +} diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index 4fa0f784a..8cd4c681c 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1,6 +1,7 @@ package sbpf import ( + "errors" "fmt" "math" "math/bits" @@ -46,6 +47,31 @@ type Interpreter struct { sbpfVersion sbpfver.SbpfVersion programId solana.PublicKey txSignature solana.Signature + + // callTargets[pc] is the resolved internal-function target of the `call imm` + // at pc (or -1). Computed once per Program at load time. + callTargets []int64 + + // Fast-path translation table indexed by (vaddr >> 32); see fastmem.go. + regions [numFastRegions]memRegion + // dirtyLo/dirtyHi: byte range written through translateInternal; + // memRegion.dirty: 4 KiB page bitmap of writes through the fast path. + dirtyLo [numFastRegions]uint64 + dirtyHi [numFastRegions]uint64 +} + +// dirtyRange returns the union of the byte ranges that may have been written +// in window idx (stack or heap), as [lo, hi). +func (ip *Interpreter) dirtyRange(idx uint64, size uint64) (lo, hi uint64) { + lo, hi = ip.dirtyLo[idx], ip.dirtyHi[idx] + if pages := ip.regions[idx].dirty; pages != 0 { + plo := uint64(bits.TrailingZeros64(pages)) << 12 + phi := uint64(bits.Len64(pages)) << 12 + lo = min(lo, plo) + hi = max(hi, phi) + } + hi = min(hi, size) + return lo, hi } type TraceSink interface { @@ -76,12 +102,13 @@ func NewInterpreter(p *Program, opts *VMOpts) *Interpreter { heap = slices.Grow(heap, opts.HeapMax-len(heap)) } heap = heap[:opts.HeapMax] - clear(heap) + // Buffers in the pool are zeroed (for their dirty range) in Finish, so + // no clear is needed here. } else { heap = newHeap() } - return &Interpreter{ + ip := &Interpreter{ textVA: p.TextVA, textBytes: p.TextBytes, text: p.Text, @@ -103,17 +130,65 @@ func NewInterpreter(p *Program, opts *VMOpts) *Interpreter { sbpfVersion: p.SbpfVersion, programId: opts.ProgramId, txSignature: opts.TxSignature, + callTargets: p.CallTargets, + } + ip.initRegions() + return ip +} + +// initRegions fills the fast-path translation table. Windows that need the +// full logic in translateInternal are left empty (rlen = wlen = 0). +func (ip *Interpreter) initRegions() { + for i := range ip.dirtyLo { + ip.dirtyLo[i] = math.MaxUint64 + ip.dirtyHi[i] = 0 + ip.regions[i].gapShift = 63 + } + if len(ip.ro) != 0 { + idx := VaddrProgram >> 32 + if ip.sbpfVersion.EnableLowerRodataVaddr() { + idx = 0 + } + ip.regions[idx] = memRegion{base: unsafe.Pointer(&ip.ro[0]), rlen: uint64(len(ip.ro)), gapShift: 63} + } + if len(ip.stack.mem) != 0 { + r := memRegion{base: unsafe.Pointer(&ip.stack.mem[0]), rlen: StackMax, wlen: StackMax, gapShift: 63} + if ip.stack.stackFrameGaps { + r.gapShift = 12 // log2(StackFrameSize) + r.gapMask = GapMask + } + ip.regions[VaddrStack>>32] = r + } + if len(ip.heap) != 0 { + ip.regions[VaddrHeap>>32] = memRegion{base: unsafe.Pointer(&ip.heap[0]), rlen: uint64(len(ip.heap)), wlen: uint64(len(ip.heap)), gapShift: 63} + } + // Finish relies on complete write tracking before returning pooled storage. + // Larger heaps (or a future larger stack) must use translateInternal's byte + // ranges: shifting the fast-path bitmap beyond page 63 silently loses writes. + // Reads remain fast; current <=256 KiB writable mappings are unchanged. + for _, idx := range []uint64{VaddrStack >> 32, VaddrHeap >> 32} { + if ip.regions[idx].wlen > fastDirtyBytes { + ip.regions[idx].wlen = 0 + } + } + if len(ip.inputRegions) == 0 && len(ip.input) != 0 { + ip.regions[VaddrInput>>32] = memRegion{base: unsafe.Pointer(&ip.input[0]), rlen: uint64(len(ip.input)), wlen: uint64(len(ip.input)), gapShift: 63} } } func (ip *Interpreter) Finish() { if UsePool { + lo, hi := ip.dirtyRange(VaddrHeap>>32, uint64(len(ip.heap))) + if hi > lo { + clear(ip.heap[lo:hi]) + } heapPool.Put(ip.heap) } + ip.stack.MarkDirty(ip.dirtyRange(VaddrStack>>32, StackMax)) ip.stack.Finish() } -func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[11]uint64) (int64, error) { +func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[16]uint64) (int64, error) { var taken bool dst := uint32(r[ins.Dst()]) src := uint32(r[ins.Src()]) @@ -200,7 +275,7 @@ func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[11]uint64) (int64, e // // This function may panic given code that doesn't pass the static verifier. func (ip *Interpreter) Run() (ret uint64, cuConsumed uint64, err error) { - var r [11]uint64 + var r [16]uint64 // 16 (not 11) so that r[ins.Dst()] (4-bit field) needs no bounds check r[1] = VaddrInput r[2] = ip.inputDataVaddr @@ -216,137 +291,228 @@ func (ip *Interpreter) Run() (ret uint64, cuConsumed uint64, err error) { // initialize pc to program entry point pc := int64(ip.entry) + // Loop-invariant state hoisted into locals so the compiler can keep them in + // registers (fields of ip may alias with the unsafe stores in the loop and + // would otherwise be reloaded on every instruction). + text := ip.text + tracing := ip.enableTracing + jmp32 := ip.sbpfVersion.EnableJmp32() + moveMem := ip.sbpfVersion.MoveMemoryInstructionClasses() + pqr := ip.sbpfVersion.EnablePqr() + staticSyscalls := ip.sbpfVersion.EnableStaticSyscalls() + callTargets := ip.callTargets + + // Instruction metering (mirrors Agave's due_insn_count / previous_instruction_meter): + // count executed instructions locally and only sync with the shared compute + // meter around syscalls and on exit. `budget` is the number of instructions + // we may still execute before the meter would be exhausted. + meter := ip.computeMeter + var budget, due uint64 + reloadBudget := func() { + if meter.Disabled() { + budget = math.MaxUint64 + } else { + budget = meter.Remaining() + } + due = 0 + // A syscall (CPI in particular) may have changed the input regions; + // drop the cached input-region fast path entry, it is re-populated on + // the next slow-path translation. (With a plain, region-less input the + // entry is static and stays.) + if len(ip.inputRegions) != 0 { + ip.regions[VaddrInput>>32] = emptyRegion + } + } + flushDue := func() { + if due != 0 { + _ = meter.Consume(due) + due = 0 + } + } + reloadBudget() + mainLoop: for i := 0; true; i++ { // Fetch - if pc < 0 || pc >= int64(len(ip.text)) { + if pc < 0 || pc >= int64(len(text)) { + flushDue() return 0, 0, &Exception{ PC: pc, Detail: fmt.Errorf("tx: %s, programId: %s - %w:", ip.txSignature, ip.programId, ExcExecutionOverrun), } } - ins := ip.getSlot(pc) - if ip.enableTracing { + ins := text[pc] + if tracing { regsDump := fmt.Sprintf("%016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x", r[0], r[1], r[2], r[3], r[4], r[5], r[6], r[7], r[8], r[9], r[10]) fmt.Printf("% 5d [%s]: %s\n", i, strings.ToUpper(regsDump), ip.disassemble(ins, 0)) } - err = ip.computeMeter.Consume(1) - if err != nil { + // Meter: identical semantics to Consume(1) before each instruction. + if due == budget { + err = cu.ErrComputeExceeded break mainLoop } + due++ // Execute - if ip.sbpfVersion.EnableJmp32() && ins.Op()&0x07 == ClassPqr { + if jmp32 && ins.Op()&0x07 == ClassPqr { pc, err = ip.executeJmp32(ins, pc, &r) goto postExecute } switch ins.Op() { case OpLdxb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint8 - v, err = ip.Read8(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 1); p != nil { + r[ins.Dst()] = uint64(*(*uint8)(p)) + } else { + var v uint8 + v, err = ip.Read8(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxh: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint16 - v, err = ip.Read16(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 2); p != nil { + r[ins.Dst()] = uint64(*(*uint16)(p)) + } else { + var v uint16 + v, err = ip.Read16(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint32 - v, err = ip.Read32(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 4); p != nil { + r[ins.Dst()] = uint64(*(*uint32)(p)) + } else { + var v uint32 + v, err = ip.Read32(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint64 - v, err = ip.Read64(vma) - r[ins.Dst()] = v + if p := ip.fastRead(vma, 8); p != nil { + r[ins.Dst()] = uint64(*(*uint64)(p)) + } else { + var v uint64 + v, err = ip.Read64(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpStb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(ins.Uimm())) + if p := ip.fastWrite(vma, 1); p != nil { + *(*uint8)(p) = uint8(ins.Uimm()) + } else { + err = ip.Write8(vma, uint8(ins.Uimm())) + } pc++ case OpSth: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(ins.Uimm())) + if p := ip.fastWrite(vma, 2); p != nil { + *(*uint16)(p) = uint16(ins.Uimm()) + } else { + err = ip.Write16(vma, uint16(ins.Uimm())) + } pc++ case OpStw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, ins.Uimm()) + if p := ip.fastWrite(vma, 4); p != nil { + *(*uint32)(p) = uint32(ins.Uimm()) + } else { + err = ip.Write32(vma, uint32(ins.Uimm())) + } pc++ case OpStdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, uint64(ins.Imm())) + if p := ip.fastWrite(vma, 8); p != nil { + *(*uint64)(p) = uint64(ins.Imm()) + } else { + err = ip.Write64(vma, uint64(ins.Imm())) + } pc++ case OpStxb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(r[ins.Src()])) + if p := ip.fastWrite(vma, 1); p != nil { + *(*uint8)(p) = uint8(r[ins.Src()]) + } else { + err = ip.Write8(vma, uint8(r[ins.Src()])) + } pc++ case OpStxh: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(r[ins.Src()])) + if p := ip.fastWrite(vma, 2); p != nil { + *(*uint16)(p) = uint16(r[ins.Src()]) + } else { + err = ip.Write16(vma, uint16(r[ins.Src()])) + } pc++ case OpStxw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, uint32(r[ins.Src()])) + if p := ip.fastWrite(vma, 4); p != nil { + *(*uint32)(p) = uint32(r[ins.Src()]) + } else { + err = ip.Write32(vma, uint32(r[ins.Src()])) + } pc++ case OpStxdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, r[ins.Src()]) + if p := ip.fastWrite(vma, 8); p != nil { + *(*uint64)(p) = uint64(r[ins.Src()]) + } else { + err = ip.Write64(vma, uint64(r[ins.Src()])) + } pc++ case OpAdd32Imm: r[ins.Dst()] = ip.signExtension(int32(r[ins.Dst()]) + ins.Imm()) @@ -383,343 +549,6 @@ mainLoop: case OpMul32Imm: r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * ins.Imm()) pc++ - case OpMul32Reg: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * int32(r[ins.Src()])) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd1BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint8 - v, err = ip.Read8(vma) - r[ins.Dst()] = uint64(v) - pc++ - } - case OpMul64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] *= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt1BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(ins.Uimm())) - pc++ - } - case OpMul64Reg: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] *= r[ins.Src()] - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt1BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(r[ins.Src()])) - pc++ - } - case OpDiv32Imm: - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) - pc++ - case OpDiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd2BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint16 - v, err = ip.Read16(vma) - r[ins.Dst()] = uint64(v) - pc++ - } - case OpLd4BReg: - if !ip.sbpfVersion.MoveMemoryInstructionClasses() { - err = ExcInvalidInstr - break - } - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint32 - v, err = ip.Read32(vma) - r[ins.Dst()] = uint64(v) - pc++ - case OpDiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] /= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt2BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(ins.Uimm())) - pc++ - } - case OpDiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] /= src - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt2BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(r[ins.Src()])) - pc++ - } - case OpSt4BReg: - if !ip.sbpfVersion.MoveMemoryInstructionClasses() { - err = ExcInvalidInstr - break - } - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, uint32(r[ins.Src()])) - pc++ - case OpLmul32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * ins.Uimm()) - pc++ - case OpLmul32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * uint32(r[ins.Src()])) - pc++ - case OpLmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] *= uint64(int64(ins.Imm())) - pc++ - case OpLmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] *= r[ins.Src()] - pc++ - case OpUhmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Uint128FromUint64(r[ins.Dst()]) - imm128 := wide.Uint128FromUint64(uint64(ins.Uimm())) - r[ins.Dst()] = dst128.Mul(imm128).RShiftN(64).Uint64() - pc++ - case OpUhmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Uint128FromUint64(r[ins.Dst()]) - regSrc128 := wide.Uint128FromUint64(r[ins.Src()]) - r[ins.Dst()] = dst128.Mul(regSrc128).RShiftN(64).Uint64() - pc++ - case OpShmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) - imm128 := wide.Int128FromInt64(int64(ins.Imm())) - r[ins.Dst()] = dst128.Mul(imm128).Uint128().RShiftN(64).Uint64() - pc++ - case OpShmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) - src128 := wide.Int128FromInt64(int64(r[ins.Src()])) - r[ins.Dst()] = dst128.Mul(src128).Uint128().RShiftN(64).Uint64() - pc++ - case OpUdiv32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) - pc++ - case OpUdiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - } - pc++ - case OpUdiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] /= uint64(ins.Uimm()) - pc++ - case OpUdiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] /= src - } else { - err = ExcDivideByZero - } - pc++ - case OpUrem32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) - pc++ - case OpUrem32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] = uint64(r[ins.Dst()] % src) - } else { - err = ExcDivideByZero - } - pc++ - case OpUrem64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] %= uint64(ins.Uimm()) - pc++ - case OpUrem64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] %= r[ins.Src()] - } else { - err = ExcDivideByZero - } - pc++ - case OpSdiv32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / ins.Imm())) - pc++ - case OpSdiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int32(r[ins.Src()]); src != 0 { - if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / src)) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSdiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / int64(ins.Imm())) - pc++ - case OpSdiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int64(r[ins.Src()]); src != 0 { - if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSrem32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % ins.Imm())) - pc++ - case OpSrem32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int32(r[ins.Src()]); src != 0 { - if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % int32(r[ins.Src()]))) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSrem64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(ins.Imm())) - pc++ - case OpSrem64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int64(r[ins.Src()]); src != 0 { - if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(r[ins.Src()])) - } else { - err = ExcDivideByZero - break - } - pc++ case OpOr32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) | ins.Uimm()) pc++ @@ -768,66 +597,6 @@ mainLoop: case OpRsh64Reg: r[ins.Dst()] >>= r[ins.Src()] & 0x3f pc++ - case OpNeg32: - if ip.sbpfVersion.DisableNeg() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(-int32(r[ins.Dst()])) - pc++ - case OpNeg64: - if !ip.sbpfVersion.DisableNeg() { - r[ins.Dst()] = uint64(-int64(r[ins.Dst()])) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt4BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, ins.Uimm()) - pc++ - } - case OpMod32Imm: - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) - pc++ - case OpMod32Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % src) - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd8BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint64 - v, err = ip.Read64(vma) - r[ins.Dst()] = v - pc++ - } - case OpMod64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] %= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt8BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, uint64(ins.Imm())) - pc++ - } - case OpMod64Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] %= src - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt8BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, r[ins.Src()]) - pc++ - } case OpXor32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) ^ ins.Uimm()) pc++ @@ -856,53 +625,6 @@ mainLoop: case OpMov64Reg: r[ins.Dst()] = r[ins.Src()] pc++ - case OpArsh32Imm: - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> ins.Uimm())) - pc++ - case OpArsh32Reg: - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> uint32(r[ins.Src()]))) - pc++ - case OpArsh64Imm: - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> ins.Imm()) - pc++ - case OpArsh64Reg: - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> (r[ins.Src()])) - pc++ - case OpHor64Imm: - if !ip.sbpfVersion.DisableLddw() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] |= uint64(ins.Uimm()) << 32 - pc++ - case OpLe: - if ip.sbpfVersion.DisableLe() { - err = ExcInvalidInstr - break - } - switch ins.Uimm() { - case 16: - r[ins.Dst()] &= math.MaxUint16 - case 32: - r[ins.Dst()] &= math.MaxUint32 - case 64: - r[ins.Dst()] &= math.MaxUint64 - default: - err = ExcUnsupportedInstruction - } - pc++ - case OpBe: - switch ins.Uimm() { - case 16: - r[ins.Dst()] = uint64(bits.ReverseBytes16(uint16(r[ins.Dst()]))) - case 32: - r[ins.Dst()] = uint64(bits.ReverseBytes32(uint32(r[ins.Dst()]))) - case 64: - r[ins.Dst()] = bits.ReverseBytes64(r[ins.Dst()]) - default: - err = ExcUnsupportedInstruction - } - pc++ case OpLddw: if ip.sbpfVersion.DisableLddw() { err = ExcInvalidInstr @@ -1024,14 +746,16 @@ mainLoop: } pc++ case OpCall: - if ip.sbpfVersion.EnableStaticSyscalls() { + if staticSyscalls { if ins.Src() == 0 { sc, ok := ip.syscalls(ins.Uimm()) if !ok { err = ExcCallDest{ins.Uimm()} break } + flushDue() r[0], err = sc.Invoke(ip, r[1], r[2], r[3], r[4], r[5]) + reloadBudget() if err != nil { err = ExcSyscallError{Err: err} } @@ -1042,7 +766,7 @@ mainLoop: err = ExcCallDest{uint32(targetPC)} break } - if ok := ip.stack.Push(r[:], pc+1); !ok { + if ok := ip.stack.Push(&r, pc+1); !ok { err = ExcCallDepth } pc = targetPC @@ -1051,60 +775,147 @@ mainLoop: } } else { if sc, ok := ip.syscalls(ins.Uimm()); ok { + flushDue() r[0], err = sc.Invoke(ip, r[1], r[2], r[3], r[4], r[5]) + reloadBudget() if err != nil { err = ExcSyscallError{Err: err} } pc++ - } else if target, ok := ip.funcs[ins.Uimm()]; ok { - ok = ip.stack.Push(r[:], pc+1) + } else { + var target int64 + var ok bool + if callTargets != nil { + target = callTargets[pc] + ok = target >= 0 + } else { + target, ok = ip.funcs[ins.Uimm()] + } if !ok { + err = ExcCallDest{ins.Uimm()} + break + } + if !ip.stack.Push(&r, pc+1) { err = ExcCallDepth } pc = target - } else { - err = ExcCallDest{ins.Uimm()} } } - case OpCallx: - var target uint64 - if ip.sbpfVersion.CallXUsesSrcReg() { - target = r[ins.Src()] - } else if ip.sbpfVersion.CallXUsesDstReg() { - target = r[ins.Dst()] + case OpExit: + var ok bool + pc, ok = ip.stack.Pop(&r) + if !ok { + ret = r[0] + break mainLoop + } + case OpMul32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * int32(r[ins.Src()])) + pc++ + case OpMul64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] *= uint64(ins.Imm()) + pc++ + case OpMul64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] *= r[ins.Src()] + pc++ + case OpDiv32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) } else { - target = r[ins.Uimm()] + err = ExcDivideByZero } - - if target < ip.textVA || target >= VaddrStack || target >= ip.textVA+uint64(len(ip.text)*8) { - err = NewExcBadAccess(target, 8, false, "jump out-of-bounds") + pc++ + case OpDiv64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) break } - targetPC := int64((target - ip.textVA) / 8) - if ok := ip.stack.Push(r[:], pc+1); !ok { - err = ExcCallDepth + r[ins.Dst()] /= uint64(ins.Imm()) + pc++ + case OpDiv64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) break } - pc = targetPC - case OpExit: - var ok bool - pc, ok = ip.stack.Pop(r[:]) - if !ok { - ret = r[0] - break mainLoop + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] /= src + } else { + err = ExcDivideByZero + } + pc++ + case OpMod32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % src) + } else { + err = ExcDivideByZero } + pc++ + case OpMod64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] %= uint64(ins.Imm()) + pc++ + case OpMod64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] %= src + } else { + err = ExcDivideByZero + } + pc++ + case OpArsh32Imm: + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> ins.Uimm())) + pc++ + case OpArsh32Reg: + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> uint32(r[ins.Src()]))) + pc++ + case OpArsh64Imm: + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> ins.Imm()) + pc++ + case OpArsh64Reg: + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> (r[ins.Src()])) + pc++ default: - err = ExcUnsupportedInstruction - return + pc, err = ip.executeCold(ins, pc, &r) + if err == errUnknownOpcode { + // Preserve the original behaviour for an unknown opcode: + // a bare (unwrapped) ExcUnsupportedInstruction. + flushDue() + return 0, 0, ExcUnsupportedInstruction + } } // Post execute postExecute: - if err == cu.ErrComputeExceeded { - err = ExcOutOfCU - } - if err != nil { + flushDue() + if err == cu.ErrComputeExceeded { + err = ExcOutOfCU + } exc := &Exception{ PC: pc, Detail: fmt.Errorf("tx: %s, programId: %s - %w:", ip.txSignature, ip.programId, err), @@ -1117,6 +928,9 @@ mainLoop: } } + flushDue() + // NB: when the loop exits because the meter is exhausted, err is the bare + // cu.ErrComputeExceeded (not wrapped in an Exception), as before. cuConsumed = ip.initialInstrMeter - ip.computeMeter.Remaining() return @@ -1186,7 +1000,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } - if lo+size > uint64(len(ip.ro)) { + if lo+size < lo || lo+size > uint64(len(ip.ro)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds program read") } return unsafe.Pointer(&ip.ro[lo]), nil @@ -1198,14 +1012,23 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } + if write { + off := StackMax - uint64(len(mem)) + ip.dirtyLo[VaddrStack>>32] = min(ip.dirtyLo[VaddrStack>>32], off) + ip.dirtyHi[VaddrStack>>32] = max(ip.dirtyHi[VaddrStack>>32], off+size) + } return unsafe.Pointer(&mem[0]), nil case VaddrHeap >> 32: if size == 0 { return emptySlice, nil } - if lo+size > uint64(len(ip.heap)) { + if lo+size < lo || lo+size > uint64(len(ip.heap)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds heap access") } + if write { + ip.dirtyLo[VaddrHeap>>32] = min(ip.dirtyLo[VaddrHeap>>32], lo) + ip.dirtyHi[VaddrHeap>>32] = max(ip.dirtyHi[VaddrHeap>>32], lo+size) + } return unsafe.Pointer(&ip.heap[lo]), nil case VaddrInput >> 32: if size == 0 { @@ -1214,7 +1037,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if len(ip.inputRegions) != 0 { return ip.translateInputRegion(lo, size, write) } - if lo+size > uint64(len(ip.input)) { + if lo+size < lo || lo+size > uint64(len(ip.input)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds input access") } return unsafe.Pointer(&ip.input[lo]), nil @@ -1255,6 +1078,8 @@ func (ip *Interpreter) translateInputRegion(offset, size uint64, write bool) (un return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } if write && (!region.Writable || requestedLen > region.RegionSize) && region.OnWrite != nil { + // The callback may replace region.Data / grow the region: drop the cache. + ip.regions[VaddrInput>>32] = emptyRegion if err := region.OnWrite(region, requestedLen); err != nil { return nil, err } @@ -1263,23 +1088,43 @@ func (ip *Interpreter) translateInputRegion(offset, size uint64, write bool) (un if !write || !region.Writable { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } + ip.regions[VaddrInput>>32] = emptyRegion region.RegionSize = region.AddressSpaceReserved } if write && !region.Writable { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "write to readonly input region") } + var base unsafe.Pointer if region.Data != nil { if requestedLen > uint64(len(region.Data)) { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } - return unsafe.Pointer(®ion.Data[regionOffset]), nil + base = unsafe.Pointer(unsafe.SliceData(region.Data)) + } else { + hostOffset := region.HostOffset + regionOffset + if hostOffset < region.HostOffset || hostOffset+size < hostOffset || hostOffset+size > uint64(len(ip.input)) { + return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") + } + base = unsafe.Pointer(&ip.input[region.HostOffset]) } - - hostOffset := region.HostOffset + regionOffset - if hostOffset < region.HostOffset || hostOffset+size < hostOffset || hostOffset+size > uint64(len(ip.input)) { - return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") + // Cache this region for the interpreter's fast path (one-entry cache, + // same idea as Agave's MappingCache). Only the currently mapped + // RegionSize bytes are exposed; anything beyond takes the slow path + // again so that OnWrite / growth semantics are preserved. + cacheLen := region.RegionSize + if region.Data == nil { + cacheLen = min(cacheLen, uint64(len(ip.input))-region.HostOffset) + } else { + cacheLen = min(cacheLen, uint64(len(region.Data))) } - return unsafe.Pointer(&ip.input[hostOffset]), nil + if cacheLen != 0 { + cached := memRegion{base: base, start: region.Offset, rlen: cacheLen, gapShift: 63} + if region.Writable { + cached.wlen = cacheLen + } + ip.regions[VaddrInput>>32] = cached + } + return unsafe.Add(base, regionOffset), nil } func (ip *Interpreter) TranslateInput(addr uint64, size uint64) ([]byte, error) { @@ -1335,6 +1180,7 @@ func (ip *Interpreter) SetInputRegionData(addr uint64, data []byte, length uint6 } region.RegionSize = length region.Writable = writable + ip.regions[VaddrInput>>32] = emptyRegion return true } @@ -1461,3 +1307,440 @@ func (ip *Interpreter) Write64(addr uint64, x uint64) error { *(*uint64)(ptr) = x return nil } + +// executeCold handles the less frequently executed opcodes. Keeping them out of +// Run keeps that function below the compiler's "big function" threshold so the +// hot helpers (metering, fast memory translation, stack push/pop) stay inlinable. +func (ip *Interpreter) executeCold(ins Slot, pc int64, r *[16]uint64) (int64, error) { + var err error + switch ins.Op() { + // In v2 these encodings are memory operations, not MUL/DIV/MOD. + // Run dispatches their non-v2 arithmetic forms on the hot path. + case OpLd1BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint8 + v, err = ip.Read8(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt1BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write8(vma, uint8(ins.Uimm())) + pc++ + case OpSt1BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write8(vma, uint8(r[ins.Src()])) + pc++ + case OpLd2BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint16 + v, err = ip.Read16(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt2BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write16(vma, uint16(ins.Uimm())) + pc++ + case OpSt2BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write16(vma, uint16(r[ins.Src()])) + pc++ + case OpLd8BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint64 + v, err = ip.Read64(vma) + r[ins.Dst()] = v + pc++ + case OpSt8BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write64(vma, uint64(ins.Imm())) + pc++ + case OpSt8BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write64(vma, r[ins.Src()]) + pc++ + case OpDiv32Imm: + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) + pc++ + case OpLd4BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint32 + v, err = ip.Read32(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt4BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write32(vma, uint32(r[ins.Src()])) + pc++ + case OpLmul32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * ins.Uimm()) + pc++ + case OpLmul32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * uint32(r[ins.Src()])) + pc++ + case OpLmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] *= uint64(int64(ins.Imm())) + pc++ + case OpLmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] *= r[ins.Src()] + pc++ + case OpUhmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Uint128FromUint64(r[ins.Dst()]) + imm128 := wide.Uint128FromUint64(uint64(ins.Uimm())) + r[ins.Dst()] = dst128.Mul(imm128).RShiftN(64).Uint64() + pc++ + case OpUhmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Uint128FromUint64(r[ins.Dst()]) + regSrc128 := wide.Uint128FromUint64(r[ins.Src()]) + r[ins.Dst()] = dst128.Mul(regSrc128).RShiftN(64).Uint64() + pc++ + case OpShmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) + imm128 := wide.Int128FromInt64(int64(ins.Imm())) + r[ins.Dst()] = dst128.Mul(imm128).Uint128().RShiftN(64).Uint64() + pc++ + case OpShmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) + src128 := wide.Int128FromInt64(int64(r[ins.Src()])) + r[ins.Dst()] = dst128.Mul(src128).Uint128().RShiftN(64).Uint64() + pc++ + case OpUdiv32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) + pc++ + case OpUdiv32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) + } else { + err = ExcDivideByZero + } + pc++ + case OpUdiv64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] /= uint64(ins.Uimm()) + pc++ + case OpUdiv64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] /= src + } else { + err = ExcDivideByZero + } + pc++ + case OpUrem32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) + pc++ + case OpUrem32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] = uint64(r[ins.Dst()] % src) + } else { + err = ExcDivideByZero + } + pc++ + case OpUrem64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] %= uint64(ins.Uimm()) + pc++ + case OpUrem64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] %= r[ins.Src()] + } else { + err = ExcDivideByZero + } + pc++ + case OpSdiv32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / ins.Imm())) + pc++ + case OpSdiv32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int32(r[ins.Src()]); src != 0 { + if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / src)) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSdiv64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / int64(ins.Imm())) + pc++ + case OpSdiv64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int64(r[ins.Src()]); src != 0 { + if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / src) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSrem32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % ins.Imm())) + pc++ + case OpSrem32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int32(r[ins.Src()]); src != 0 { + if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % int32(r[ins.Src()]))) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSrem64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(ins.Imm())) + pc++ + case OpSrem64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int64(r[ins.Src()]); src != 0 { + if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(r[ins.Src()])) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpNeg32: + if ip.sbpfVersion.DisableNeg() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(-int32(r[ins.Dst()])) + pc++ + case OpNeg64: + if !ip.sbpfVersion.DisableNeg() { + r[ins.Dst()] = uint64(-int64(r[ins.Dst()])) + pc++ + } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { + // OpSt4BImm + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write32(vma, ins.Uimm()) + pc++ + } + case OpMod32Imm: + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) + pc++ + case OpHor64Imm: + if !ip.sbpfVersion.DisableLddw() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] |= uint64(ins.Uimm()) << 32 + pc++ + case OpLe: + if ip.sbpfVersion.DisableLe() { + err = ExcInvalidInstr + break + } + switch ins.Uimm() { + case 16: + r[ins.Dst()] &= math.MaxUint16 + case 32: + r[ins.Dst()] &= math.MaxUint32 + case 64: + r[ins.Dst()] &= math.MaxUint64 + default: + err = ExcUnsupportedInstruction + } + pc++ + case OpBe: + switch ins.Uimm() { + case 16: + r[ins.Dst()] = uint64(bits.ReverseBytes16(uint16(r[ins.Dst()]))) + case 32: + r[ins.Dst()] = uint64(bits.ReverseBytes32(uint32(r[ins.Dst()]))) + case 64: + r[ins.Dst()] = bits.ReverseBytes64(r[ins.Dst()]) + default: + err = ExcUnsupportedInstruction + } + pc++ + case OpCallx: + var target uint64 + if ip.sbpfVersion.CallXUsesSrcReg() { + target = r[ins.Src()] + } else if ip.sbpfVersion.CallXUsesDstReg() { + target = r[ins.Dst()] + } else { + target = r[ins.Uimm()] + } + + if target < ip.textVA || target >= VaddrStack || target >= ip.textVA+uint64(len(ip.text)*8) { + err = NewExcBadAccess(target, 8, false, "jump out-of-bounds") + break + } + targetPC := int64((target - ip.textVA) / 8) + if ok := ip.stack.Push(r, pc+1); !ok { + err = ExcCallDepth + break + } + pc = targetPC + default: + err = errUnknownOpcode + } + return pc, err +} + +// errUnknownOpcode is an internal sentinel returned by executeCold for an +// opcode that is not handled by either switch; Run turns it into the bare +// ExcUnsupportedInstruction return of the original implementation. +var errUnknownOpcode = errors.New("unknown opcode") diff --git a/pkg/sbpf/interpreter_v2_test.go b/pkg/sbpf/interpreter_v2_test.go new file mode 100644 index 000000000..39fd1e161 --- /dev/null +++ b/pkg/sbpf/interpreter_v2_test.go @@ -0,0 +1,113 @@ +package sbpf + +import ( + "bytes" + "encoding/binary" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" + "github.com/stretchr/testify/require" +) + +// These byte values alias arithmetic instructions outside v2. Exercise all +// relocated memory instructions through Run, not just the cold handler. +func TestInterpreterV2MemoryOpcodes(t *testing.T) { + for _, width := range []int{1, 2, 4, 8} { + loads := map[int]uint8{1: OpLd1BReg, 2: OpLd2BReg, 4: OpLd4BReg, 8: OpLd8BReg} + immediates := map[int]uint8{1: OpSt1BImm, 2: OpSt2BImm, 4: OpSt4BImm, 8: OpSt8BImm} + registers := map[int]uint8{1: OpSt1BReg, 2: OpSt2BReg, 4: OpSt4BReg, 8: OpSt8BReg} + for _, kind := range []string{"load", "store_imm", "store_reg"} { + for _, region := range []string{"heap", "stack", "input", "rodata", "unmapped"} { + t.Run(fmt.Sprintf("%s/%d/%s", kind, width, region), func(t *testing.T) { + const value = uint64(0xfedcba9876543210) + immediate := uint32(0xf123abcd) + var addr uint64 + switch region { + case "heap": + addr = VaddrHeap + 9 + case "stack": + addr = VaddrStack + 9 + case "input": + addr = VaddrInput + 9 + case "rodata": + addr = VaddrProgram + 9 + case "unmapped": + addr = 0x600000009 + } + // Negative offsets and unaligned addresses must behave identically to + // the original interpreter, including the post-instruction exception PC. + text := diffLoadImm64(5, addr+3, sbpfver.SbpfVersionV2) + text = append(text, diffLoadImm64(6, value, sbpfver.SbpfVersionV2)...) + var op Slot + switch kind { + case "load": + op = slot(loads[width], 0, 5, -3, 0) + case "store_imm": + op = slot(immediates[width], 5, 0, -3, immediate) + case "store_reg": + op = slot(registers[width], 5, 6, -3, 0) + } + text = append(text, op, slot(OpExit, 0, 0, 0, 0)) + p := mkProgram(text, sbpfver.SbpfVersionV2) + p.RO = bytes.Repeat([]byte{0xa5}, 32) + require.NoError(t, p.Verify()) + cm := cu.NewComputeMeter(100) + ip := NewInterpreter(p, &VMOpts{HeapMax: 32, Input: bytes.Repeat([]byte{0xa5}, 32), ComputeMeter: &cm, Syscalls: noSyscalls}) + defer ip.Finish() + var memory []byte + switch region { + case "heap": + memory = ip.heap + case "stack": + memory = ip.stack.mem + case "input": + memory = ip.input + case "rodata": + memory = p.RO + } + if memory != nil { + // Use Write for writable VM storage so pooled-memory tracking is kept. + if region != "rodata" { + require.NoError(t, ip.Write(addr-9, bytes.Repeat([]byte{0xa5}, 32))) + } + } + before := append([]byte(nil), memory...) + ret, used, err := ip.Run() + if region == "unmapped" || (region == "rodata" && kind != "load") { + require.Error(t, err) + var exc *Exception + require.ErrorAs(t, err, &exc) + require.Equal(t, int64(5), exc.PC) + var access ExcBadAccess + require.ErrorAs(t, err, &access) + require.Equal(t, addr, access.Addr) + require.Equal(t, uint64(width), access.Size) + require.Equal(t, kind != "load", access.Write) + require.Equal(t, uint64(95), cm.Remaining()) + require.Equal(t, before, memory) + return + } + require.NoError(t, err) + require.Equal(t, uint64(6), used) + require.Equal(t, uint64(94), cm.Remaining()) + var encoded [8]byte + if kind == "load" { + copy(encoded[:], before[9:9+width]) + require.Equal(t, binary.LittleEndian.Uint64(encoded[:]), ret) + } else { + v := value + if kind == "store_imm" { + signed := int32(immediate) + v = uint64(int64(signed)) + } + binary.LittleEndian.PutUint64(encoded[:], v) + copy(before[9:9+width], encoded[:width]) + } + require.Equal(t, before, memory) + }) + } + } + } +} diff --git a/pkg/sbpf/loader/loader.go b/pkg/sbpf/loader/loader.go index 7b9dc52ca..23a26ec76 100644 --- a/pkg/sbpf/loader/loader.go +++ b/pkg/sbpf/loader/loader.go @@ -162,7 +162,7 @@ func parseSlots(bs []byte) []sbpf.Slot { } func (l *Loader) getProgram() *sbpf.Program { - return &sbpf.Program{ + p := &sbpf.Program{ RO: l.program, TextBytes: l.text, Text: parseSlots(l.text), @@ -171,4 +171,6 @@ func (l *Loader) getProgram() *sbpf.Program { Funcs: l.funcs, SbpfVersion: l.sbpfVersion(), } + p.ResolveCallTargets() + return p } diff --git a/pkg/sbpf/loader/token_perf_bench_test.go b/pkg/sbpf/loader/token_perf_bench_test.go new file mode 100644 index 000000000..c2534e586 --- /dev/null +++ b/pkg/sbpf/loader/token_perf_bench_test.go @@ -0,0 +1,425 @@ +package loader_test + +import ( + "encoding/binary" + "errors" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/Overclock-Validator/mithril/pkg/sbpf/loader" +) + +// ---- minimal syscall set mirroring pkg/sealevel semantics (CU costs + memory behaviour) ---- + +const ( + cuSyscallBase = 100 + cuMemOpBase = 10 + cuCpiBytesPerCU = 250 +) + +type stats struct { + logs int + memcpy int + memset int + memcmp int + memmove int + bytes uint64 +} + +var st stats + +func memOpConsume(vm sbpf.VM, n uint64) error { + cost := max(uint64(cuMemOpBase), n/cuCpiBytesPerCU) + return vm.ComputeMeter().Consume(cost) +} + +func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) (err error) { + srcBuf := make([]byte, n) // same allocation pattern as sealevel/syscalls_mem.go + err = vm.Read(src, srcBuf) + if err != nil { + return + } + err = vm.Write(dst, srcBuf) + return +} + +func isNonOverlapping(src, srcLen, dst, dstLen uint64) bool { + if src > dst { + return src-dst >= dstLen + } + return dst-src >= srcLen +} + +var syscallMemcpy = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, src, n uint64) (uint64, error) { + st.memcpy++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + if !isNonOverlapping(src, n, dst, n) { + return 0, errors.New("overlapping") + } + if n == 0 { + return 0, nil + } + return 0, memmoveImplInternal(vm, dst, src, n) +}) + +var syscallMemmove = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, src, n uint64) (uint64, error) { + st.memmove++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + return 0, memmoveImplInternal(vm, dst, src, n) +}) + +var syscallMemcmp = sbpf.SyscallFunc4(func(vm sbpf.VM, a1, a2, n, res uint64) (uint64, error) { + st.memcmp++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + s1, err := vm.Translate(a1, n, false) + if err != nil { + return 0, err + } + s2, err := vm.Translate(a2, n, false) + if err != nil { + return 0, err + } + r := int32(0) + for i := uint64(0); i < n; i++ { + if s1[i] != s2[i] { + r = int32(s1[i]) - int32(s2[i]) + break + } + } + out, err := vm.Translate(res, 4, true) + if err != nil { + return 0, err + } + binary.LittleEndian.PutUint32(out, uint32(r)) + return 0, nil +}) + +var syscallMemset = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, c, n uint64) (uint64, error) { + st.memset++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + mem, err := vm.Translate(dst, n, true) + if err != nil { + return 0, err + } + for i := uint64(0); i < n; i++ { + mem[i] = byte(c) + } + return 0, nil +}) + +var syscallLog = sbpf.SyscallFunc2(func(vm sbpf.VM, ptr, strlen uint64) (uint64, error) { + st.logs++ + if err := vm.ComputeMeter().Consume(max(uint64(cuSyscallBase), strlen)); err != nil { + return 0, err + } + buf := make([]byte, strlen) + if err := vm.Read(ptr, buf); err != nil { + return 0, err + } + _ = string(buf) + return 0, nil +}) + +var syscallLog64 = sbpf.SyscallFunc5(func(vm sbpf.VM, a, b, c, d, e uint64) (uint64, error) { + st.logs++ + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallLogPubkey = sbpf.SyscallFunc1(func(vm sbpf.VM, a uint64) (uint64, error) { + st.logs++ + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallLogCUs = sbpf.SyscallFunc0(func(vm sbpf.VM) (uint64, error) { + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallAbort = sbpf.SyscallFunc0(func(vm sbpf.VM) (uint64, error) { return 0, errors.New("abort") }) +var syscallPanic = sbpf.SyscallFunc4(func(vm sbpf.VM, f, l, line, col uint64) (uint64, error) { + return 0, errors.New("panic") +}) +var syscallAllocFree = sbpf.SyscallFunc2(func(vm sbpf.VM, size, free uint64) (uint64, error) { + if free != 0 { + return 0, nil + } + hs := (vm.HeapSize() + 7) &^ 7 + addr := sbpf.VaddrHeap + hs + hs += size + if hs > vm.HeapMax() { + return 0, nil + } + vm.UpdateHeapSize(hs) + return addr, nil +}) + +var registry = map[uint32]sbpf.Syscall{ + sbpf.SymbolHash("abort"): syscallAbort, + sbpf.SymbolHash("sol_panic_"): syscallPanic, + sbpf.SymbolHash("sol_log_"): syscallLog, + sbpf.SymbolHash("sol_log_64_"): syscallLog64, + sbpf.SymbolHash("sol_log_pubkey"): syscallLogPubkey, + sbpf.SymbolHash("sol_log_compute_units_"): syscallLogCUs, + sbpf.SymbolHash("sol_memcpy_"): syscallMemcpy, + sbpf.SymbolHash("sol_memmove_"): syscallMemmove, + sbpf.SymbolHash("sol_memcmp_"): syscallMemcmp, + sbpf.SymbolHash("sol_memset_"): syscallMemset, + sbpf.SymbolHash("sol_alloc_free_"): syscallAllocFree, +} + +var syscalls = sbpf.SyscallRegistry(func(h uint32) (sbpf.Syscall, bool) { + s, ok := registry[h] + return s, ok +}) + +// ---- aligned input serialization (BPF loader v2/v3 format, no direct mapping) ---- + +const maxPermittedDataIncrease = 10 * 1024 + +type acct struct { + key, owner [32]byte + lamports uint64 + data []byte + signer, writable bool +} + +func serializeAligned(accts []acct, instrData []byte, programId [32]byte) []byte { + out := binary.LittleEndian.AppendUint64(nil, uint64(len(accts))) + for _, a := range accts { + out = append(out, 0xff) + out = append(out, b2u8(a.signer), b2u8(a.writable), 0) + out = append(out, 0, 0, 0, 0) // original_data_len + out = append(out, a.key[:]...) + out = append(out, a.owner[:]...) + out = binary.LittleEndian.AppendUint64(out, a.lamports) + out = binary.LittleEndian.AppendUint64(out, uint64(len(a.data))) + out = append(out, a.data...) + pad := maxPermittedDataIncrease + ((8 - len(a.data)%8) % 8) + out = append(out, make([]byte, pad)...) + out = binary.LittleEndian.AppendUint64(out, ^uint64(0)) // rent epoch + } + out = binary.LittleEndian.AppendUint64(out, uint64(len(instrData))) + out = append(out, instrData...) + out = append(out, programId[:]...) + return out +} + +func b2u8(b bool) byte { + if b { + return 1 + } + return 0 +} + +// SPL token account layout (165 bytes) +func tokenAccount(mint, owner [32]byte, amount uint64) []byte { + d := make([]byte, 165) + copy(d[0:32], mint[:]) + copy(d[32:64], owner[:]) + binary.LittleEndian.PutUint64(d[64:72], amount) + // delegate: COption none (4 bytes 0) + 32 + d[108] = 1 // state = Initialized + // is_native COption none, delegated_amount 0, close_authority none + return d +} + +func key(b byte) [32]byte { + var k [32]byte + for i := range k { + k[i] = b + } + return k +} + +func loadTokenProgram(tb testing.TB) *sbpf.Program { + elfBytes := fixtures.Load(tb, "sbpf", "spl-token.so") + f := features.NewFeaturesDefault() + l, err := loader.NewLoaderWithSyscalls(elfBytes, syscalls, false, f) + if err != nil { + tb.Fatal(err) + } + p, err := l.Load() + if err != nil { + tb.Fatal(err) + } + if err := p.Verify(); err != nil { + tb.Fatal(err) + } + return p +} + +func transferInput(programId [32]byte) ([]byte, []acct) { + mint := key(0x11) + authority := key(0x22) + src := key(0x33) + dst := key(0x44) + accts := []acct{ + {key: src, owner: programId, lamports: 2039280, data: tokenAccount(mint, authority, 1_000_000), writable: true}, + {key: dst, owner: programId, lamports: 2039280, data: tokenAccount(mint, key(0x55), 5), writable: true}, + {key: authority, owner: key(0), lamports: 1_000_000_000, data: nil, signer: true}, + } + instr := append([]byte{3}, binary.LittleEndian.AppendUint64(nil, 1000)...) + return serializeAligned(accts, instr, programId), accts +} + +func runTransfer(tb testing.TB, p *sbpf.Program, input []byte) (uint64, uint64) { + cm := cu.NewComputeMeter(200_000) + ip := sbpf.NewInterpreter(p, &sbpf.VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + }) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + tb.Fatal(err) + } + return ret, used +} + +func TestTokenTransfer(t *testing.T) { + p := loadTokenProgram(t) + programId := key(0x99) + input, _ := transferInput(programId) + st = stats{} + ret, used := runTransfer(t, p, input) + t.Logf("ret=%d cuUsed=%d stats=%+v", ret, used, st) + // verify balances changed in the serialized input + // account 0 data starts at 8 + 8 + 32+32+8+8 = 96 + srcAmt := binary.LittleEndian.Uint64(input[96+64:]) + off1 := 8 + (8 + 32 + 32 + 8 + 8 + 165 + maxPermittedDataIncrease + 3 + 8) + dstAmt := binary.LittleEndian.Uint64(input[off1+88+64:]) + t.Logf("src=%d dst=%d", srcAmt, dstAmt) + if ret != 0 || srcAmt != 999_000 || dstAmt != 1005 { + t.Fatalf("unexpected result ret=%d src=%d dst=%d", ret, srcAmt, dstAmt) + } +} + +func BenchmarkTokenTransfer(b *testing.B) { + p := loadTokenProgram(b) + programId := key(0x99) + input, _ := transferInput(programId) + orig := append([]byte(nil), input...) + b.ReportAllocs() + b.ResetTimer() + var used uint64 + for i := 0; i < b.N; i++ { + copy(input, orig) + _, used = runTransfer(b, p, input) + } + b.StopTimer() + b.ReportMetric(float64(used), "cu/op") + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N)/float64(used), "ns/cu") +} + +func BenchmarkTokenLoadVerify(b *testing.B) { + elfBytes := fixtures.Load(b, "sbpf", "spl-token.so") + f := features.NewFeaturesDefault() + b.ReportAllocs() + for i := 0; i < b.N; i++ { + l, err := loader.NewLoaderWithSyscalls(elfBytes, syscalls, false, f) + if err != nil { + b.Fatal(err) + } + p, err := l.Load() + if err != nil { + b.Fatal(err) + } + if err := p.Verify(); err != nil { + b.Fatal(err) + } + } +} + +// ---- VASA layout (VirtualAddressSpaceAdjustments active, direct mapping off) ---- +// Mirrors serializeParametersAligned with vasa=true, directMapping=false: +// same bytes as the aligned layout, but the input window is split into +// metadata regions and per-account data regions. + +func vasaRegions(accts []acct, instrLen int) []sbpf.InputRegion { + var regions []sbpf.InputRegion + var regionStart, hostRegionStart uint64 + vmOff := uint64(8) + for i, a := range accts { + l := vmOff // host offset == vm offset in this layout + dataLen := uint64(len(a.data)) + align := (8 - dataLen%8) % 8 + reserved := dataLen + maxPermittedDataIncrease + dataStart := vmOff + 88 + if dataStart > regionStart { + regions = append(regions, sbpf.InputRegion{Offset: regionStart, HostOffset: hostRegionStart, + RegionSize: dataStart - regionStart, AddressSpaceReserved: dataStart - regionStart, Writable: true, AccountIndex: -1}) + } + regions = append(regions, sbpf.InputRegion{Offset: dataStart, HostOffset: l + 88, RegionSize: dataLen, + AddressSpaceReserved: reserved, Writable: a.writable, AccountIndex: i}) + hostRegionStart = l + 88 + reserved + regionStart = dataStart + reserved + vmOff += 88 + reserved + align + 8 + } + end := vmOff + 8 + uint64(instrLen) + 32 + regions = append(regions, sbpf.InputRegion{Offset: regionStart, HostOffset: hostRegionStart, + RegionSize: end - regionStart, AddressSpaceReserved: end - regionStart, Writable: true, AccountIndex: -1}) + return regions +} + +func runTransferVasa(tb testing.TB, p *sbpf.Program, input []byte, regions []sbpf.InputRegion) (uint64, uint64) { + cm := cu.NewComputeMeter(200_000) + // regions are mutated by the VM (RegionSize on growth), so copy per run + rc := append([]sbpf.InputRegion(nil), regions...) + ip := sbpf.NewInterpreter(p, &sbpf.VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + InputRegions: rc, + DisableStackFrameGaps: true, + }) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + tb.Fatal(err) + } + return ret, used +} + +func TestTokenTransferVasa(t *testing.T) { + p := loadTokenProgram(t) + programId := key(0x99) + input, accts := transferInput(programId) + regions := vasaRegions(accts, 9) + if regions[len(regions)-1].Offset+regions[len(regions)-1].RegionSize != uint64(len(input)) { + t.Fatalf("region layout mismatch: %d vs %d", regions[len(regions)-1].Offset+regions[len(regions)-1].RegionSize, len(input)) + } + ret, used := runTransferVasa(t, p, input, regions) + srcAmt := binary.LittleEndian.Uint64(input[96+64:]) + if ret != 0 || srcAmt != 999_000 { + t.Fatalf("unexpected ret=%d src=%d", ret, srcAmt) + } + t.Logf("ret=%d cu=%d regions=%d", ret, used, len(regions)) +} + +func BenchmarkTokenTransferVasa(b *testing.B) { + p := loadTokenProgram(b) + programId := key(0x99) + input, accts := transferInput(programId) + regions := vasaRegions(accts, 9) + orig := append([]byte(nil), input...) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + copy(input, orig) + runTransferVasa(b, p, input, regions) + } +} diff --git a/pkg/sbpf/perf_bench_test.go b/pkg/sbpf/perf_bench_test.go new file mode 100644 index 000000000..1d34d4d7a --- /dev/null +++ b/pkg/sbpf/perf_bench_test.go @@ -0,0 +1,193 @@ +package sbpf + +import ( + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" +) + +func slot(op uint8, dst uint8, src uint8, off int16, imm uint32) Slot { + return Slot(op) | Slot(dst)<<8 | Slot(src)<<12 | Slot(uint16(off))<<16 | Slot(imm)<<32 +} + +func slotsToBytes(slots []Slot) []byte { + out := make([]byte, len(slots)*SlotSize) + for i, s := range slots { + binary.LittleEndian.PutUint64(out[i*SlotSize:], uint64(s)) + } + return out +} + +func mkProgram(text []Slot, ver uint32) *Program { + return &Program{ + TextBytes: slotsToBytes(text), + Text: text, + TextVA: VaddrProgram, + Entrypoint: 0, + Funcs: map[uint32]int64{}, + SbpfVersion: sbpfver.SbpfVersion{Version: ver}, + } +} + +var noSyscalls = SyscallRegistry(func(uint32) (Syscall, bool) { return nil, false }) + +// resolveCallTargetsIfSupported precomputes internal call targets on trees +// that have Program.ResolveCallTargets (the loader does this at load time); +// it is a no-op on the baseline tree so the same benchmark code runs on both. +func resolveCallTargetsIfSupported(p *Program) { + if r, ok := any(p).(interface{ ResolveCallTargets() }); ok { + r.ResolveCallTargets() + } +} + +// aluLoop: r1 = N; loop: r2 += r1; r2 ^= r3; r3 = r2; r3 *= 7; r3 >>= 3; r1 -= 1; jne r1,0 loop; exit +// 7 instructions per iteration. +func aluLoopProgram(n uint32, ver uint32) *Program { + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + slot(OpMov64Imm, 3, 0, 0, 3), + // loop @3 + slot(OpAdd64Reg, 2, 1, 0, 0), + slot(OpXor64Reg, 2, 3, 0, 0), + slot(OpMov64Reg, 3, 2, 0, 0), + slot(OpMul64Imm, 3, 0, 0, 7), + slot(OpRsh64Imm, 3, 0, 0, 3), + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -7, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + return mkProgram(text, ver) +} + +// memLoop: writes and reads 8 byte values on the stack frame and heap. +// r1 = N; r4 = r10 - 4096 (frame base); r5 = heap base +// loop: stxdw [r4+0], r1; ldxdw r6, [r4+0]; add r2, r6; stxdw [r5+8], r2; ldxdw r7,[r5+8]; xor r2,r7 ; r1 -= 1; jne +func memLoopProgram(n uint32, ver uint32) *Program { + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + slot(OpMov64Reg, 4, 10, 0, 0), + slot(OpAdd64Imm, 4, 0, 0, uint32(0xfffff000)), // r4 = r10 - 4096 + slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), + slot(0, 0, 0, 0, uint32(VaddrHeap>>32)), + // loop @6 + slot(OpStxdw, 4, 1, 0, 0), + slot(OpLdxdw, 6, 4, 0, 0), + slot(OpAdd64Reg, 2, 6, 0, 0), + slot(OpStxdw, 5, 2, 8, 0), + slot(OpLdxdw, 7, 5, 8, 0), + slot(OpXor64Reg, 2, 7, 0, 0), + slot(OpStxw, 5, 2, 16, 0), + slot(OpLdxb, 8, 5, 16, 0), + slot(OpAdd64Reg, 2, 8, 0, 0), + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -11, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + return mkProgram(text, ver) +} + +// callLoop: calls a tiny function N times (tests Push/Pop + call resolution) +func callLoopProgram(n uint32, ver uint32) *Program { + fnPC := int64(7) + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + // loop @2 + slot(OpCall, 0, 0, 0, 0), // patched below + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -3, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + // fn @7 + slot(OpAdd64Imm, 2, 0, 0, 3), + slot(OpXor64Reg, 2, 1, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + p := mkProgram(text, ver) + if ver >= sbpfver.SbpfVersionV3 { + // relative call: target = pc + imm + 1 ; pc=2 -> imm = 7-2-1 = 4 + text[2] = slot(OpCall, 0, 1, 0, uint32(fnPC-2-1)) + } else { + h := PCHash(uint64(fnPC)) + p.Funcs[h] = fnPC + text[2] = slot(OpCall, 0, 0, 0, h) + } + p.TextBytes = slotsToBytes(text) + return p +} + +func runProgram(b *testing.B, p *Program, input []byte, syscalls SyscallRegistry, budget uint64) uint64 { + cm := cu.NewComputeMeter(budget) + ip := NewInterpreter(p, &VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + }) + ret, _, err := ip.Run() + ip.Finish() + if err != nil { + b.Fatal(err) + } + return ret +} + +func benchLoop(b *testing.B, p *Program, insnsPerRun uint64) { + if err := p.Verify(); err != nil { + b.Fatal(err) + } + resolveCallTargetsIfSupported(p) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + runProgram(b, p, nil, noSyscalls, 1<<40) + } + b.StopTimer() + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(uint64(b.N)*insnsPerRun), "ns/insn") +} + +const loopN = 200_000 + +func BenchmarkAluLoopV0(b *testing.B) { benchLoop(b, aluLoopProgram(loopN, 0), 3+7*loopN+2) } +func BenchmarkAluLoopV3(b *testing.B) { benchLoop(b, aluLoopProgram(loopN, 3), 3+7*loopN+2) } +func BenchmarkMemLoopV0(b *testing.B) { benchLoop(b, memLoopProgram(loopN, 0), 5+11*loopN+2) } +func BenchmarkMemLoopV3(b *testing.B) { benchLoop(b, memLoopProgram(loopN, 3), 5+11*loopN+2) } +func BenchmarkCallLoopV0(b *testing.B) { + benchLoop(b, callLoopProgram(loopN, 0), 2+6*loopN+2) +} +func BenchmarkCallLoopV3(b *testing.B) { + benchLoop(b, callLoopProgram(loopN, 3), 2+6*loopN+2) +} + +// Interpreter setup/teardown cost only (tiny program). +func BenchmarkNewInterpreterAndExit(b *testing.B) { + p := mkProgram([]Slot{slot(OpExit, 0, 0, 0, 0)}, 0) + b.ReportAllocs() + for i := 0; i < b.N; i++ { + runProgram(b, p, nil, noSyscalls, 1000) + } +} + +func TestSyntheticProgramsRun(t *testing.T) { + for _, ver := range []uint32{0, 3} { + for _, p := range []*Program{aluLoopProgram(1000, ver), memLoopProgram(1000, ver), callLoopProgram(1000, ver)} { + if err := p.Verify(); err != nil { + t.Fatal(err) + } + cm := cu.NewComputeMeter(1 << 30) + ip := NewInterpreter(p, &VMOpts{HeapMax: 32 * 1024, Syscalls: noSyscalls, ComputeMeter: &cm}) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + t.Fatal(err) + } + t.Logf("ver=%d ret=%d cu=%d", ver, ret, used) + } + } +} diff --git a/pkg/sbpf/perf_differential_test.go b/pkg/sbpf/perf_differential_test.go new file mode 100644 index 000000000..3ad824959 --- /dev/null +++ b/pkg/sbpf/perf_differential_test.go @@ -0,0 +1,425 @@ +package sbpf + +import ( + "bufio" + "crypto/sha256" + "fmt" + "hash/fnv" + "io" + "math/rand" + "os" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" +) + +// Differential test: generates deterministic pseudo-random programs and dumps +// (return value, error string, CU consumed, memory hash) per program to the +// file named by SBPF_DIFF_OUT. Running it against the baseline and the +// optimized interpreter and diffing the two files checks that observable +// behaviour is identical. + +type diffSyscall struct { + fn func(vm VM, r1, r2, r3, r4, r5 uint64) (uint64, error) +} + +func (s diffSyscall) Invoke(vm VM, r1, r2, r3, r4, r5 uint64) (uint64, error) { + return s.fn(vm, r1, r2, r3, r4, r5) +} + +var ( + hashPoke = SymbolHash("poke") // write r3 bytes of value r2 at r1 via vm.Write + hashPeek = SymbolHash("peek") // read 8 bytes at r1 -> r0 + hashCopy = SymbolHash("copy") // copy r3 bytes from r2 to r1 (Translate based) + hashBurn = SymbolHash("burn") // consume r1 CU + hashSetLen = SymbolHash("setlen") // SetInputRegionLength(r1, r2, r3!=0) +) + +func diffRegistry(h uint32) (Syscall, bool) { + switch h { + case hashPoke: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + if r3 > 4096 { + r3 = 4096 + } + buf := make([]byte, r3) + for i := range buf { + buf[i] = byte(r2 + uint64(i)) + } + return 0, vm.Write(r1, buf) + }}, true + case hashPeek: + return diffSyscall{func(vm VM, r1, _, _, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + return vm.Read64(r1) + }}, true + case hashCopy: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + if r3 > 4096 { + r3 = 4096 + } + src, err := vm.Translate(r2, r3, false) + if err != nil { + return 0, err + } + dst, err := vm.Translate(r1, r3, true) + if err != nil { + return 0, err + } + copy(dst, src) + return 0, nil + }}, true + case hashBurn: + return diffSyscall{func(vm VM, r1, _, _, _, _ uint64) (uint64, error) { + return 0, vm.ComputeMeter().Consume(r1 & 0xff) + }}, true + case hashSetLen: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + ip := vm.(*Interpreter) + ok := ip.SetInputRegionLength(r1, r2, r3 != 0) + if ok { + return 1, nil + } + return 0, nil + }}, true + } + return nil, false +} + +// v2 replaces LDDW with MOV32 + HOR64. MOV32 avoids sign-extending the +// low word before ORing in the high word. +func diffLoadImm64(dst uint8, value uint64, ver uint32) []Slot { + if ver == sbpfver.SbpfVersionV2 { + return []Slot{slot(OpMov32Imm, dst, 0, 0, uint32(value)), slot(OpHor64Imm, dst, 0, 0, uint32(value>>32))} + } + return []Slot{slot(OpLddw, dst, 0, 0, uint32(value)), slot(0, 0, 0, 0, uint32(value>>32))} +} + +func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { + reg := func() uint8 { return uint8(1 + rng.Intn(9)) } // r1..r9 + imm := func() uint32 { + switch rng.Intn(4) { + case 0: + return uint32(rng.Intn(16)) + case 1: + return uint32(int32(-rng.Intn(16))) + case 2: + return rng.Uint32() + default: + return uint32(rng.Intn(4096)) + } + } + alu64 := []uint8{OpAdd64Imm, OpAdd64Reg, OpSub64Imm, OpSub64Reg, OpMul64Imm, OpMul64Reg, OpDiv64Imm, OpDiv64Reg, + OpOr64Imm, OpOr64Reg, OpAnd64Imm, OpAnd64Reg, OpLsh64Imm, OpLsh64Reg, OpRsh64Imm, OpRsh64Reg, OpMod64Imm, OpMod64Reg, + OpXor64Imm, OpXor64Reg, OpMov64Imm, OpMov64Reg, OpArsh64Imm, OpArsh64Reg, OpNeg64, + OpAdd32Imm, OpAdd32Reg, OpSub32Imm, OpSub32Reg, OpMul32Imm, OpMul32Reg, OpDiv32Imm, OpDiv32Reg, OpOr32Imm, OpOr32Reg, + OpAnd32Imm, OpAnd32Reg, OpLsh32Imm, OpLsh32Reg, OpRsh32Imm, OpRsh32Reg, OpMod32Imm, OpMod32Reg, OpXor32Imm, OpXor32Reg, + OpMov32Imm, OpMov32Reg, OpArsh32Imm, OpArsh32Reg, OpNeg32, OpLe, OpBe} + if ver == sbpfver.SbpfVersionV2 { + // Arithmetic and memory encodings both change in v2. Keep generated ALU + // operations arithmetic rather than generating unintended memory accesses. + replacements := map[uint8]uint8{ + OpMul32Imm: OpLmul32Imm, OpMul32Reg: OpLmul32Reg, + OpMul64Imm: OpLmul64Imm, OpMul64Reg: OpLmul64Reg, + OpDiv32Imm: OpUdiv32Imm, OpDiv32Reg: OpUdiv32Reg, + OpDiv64Imm: OpUdiv64Imm, OpDiv64Reg: OpUdiv64Reg, + OpMod32Imm: OpUrem32Imm, OpMod32Reg: OpUrem32Reg, + OpMod64Imm: OpUrem64Imm, OpMod64Reg: OpUrem64Reg, + } + filtered := alu64[:0] + for _, op := range alu64 { + if op == OpNeg32 || op == OpNeg64 || op == OpLe { + continue + } + if replacement, ok := replacements[op]; ok { + op = replacement + } + filtered = append(filtered, op) + } + alu64 = filtered + } + jmp := []uint8{OpJeqImm, OpJeqReg, OpJgtImm, OpJgtReg, OpJgeImm, OpJgeReg, OpJltImm, OpJltReg, OpJleImm, OpJleReg, + OpJsetImm, OpJsetReg, OpJneImm, OpJneReg, OpJsgtImm, OpJsgtReg, OpJsgeImm, OpJsgeReg, OpJsltImm, OpJsltReg, OpJsleImm, OpJsleReg} + switch rng.Intn(10) { + case 0, 1, 2, 3: // alu + op := alu64[rng.Intn(len(alu64))] + i := imm() + if op == OpLe || op == OpBe { + i = []uint32{16, 32, 64}[rng.Intn(3)] + } + if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm || op == OpUdiv32Imm || op == OpUdiv64Imm || op == OpUrem32Imm || op == OpUrem64Imm) && i == 0 { + i = 3 + } + switch op { + case OpLsh32Imm, OpRsh32Imm, OpArsh32Imm: + i = uint32(rng.Intn(32)) + case OpLsh64Imm, OpRsh64Imm, OpArsh64Imm: + i = uint32(rng.Intn(64)) + } + return []Slot{slot(op, reg(), reg(), 0, i)} + case 4: // load + ops := []uint8{OpLdxb, OpLdxh, OpLdxw, OpLdxdw} + if ver == sbpfver.SbpfVersionV2 { + ops = []uint8{OpLd1BReg, OpLd2BReg, OpLd4BReg, OpLd8BReg} + } + // base register: r10 (stack) or r5 (heap ptr) or r1 (input ptr) or random + var base uint8 + var off int16 + switch rng.Intn(4) { + case 0: + base, off = 10, int16(-rng.Intn(4096)) + case 1: + base, off = 5, int16(rng.Intn(1024)) + case 2: + base, off = 1, int16(rng.Intn(600)) + default: + base, off = reg(), int16(rng.Intn(65536)-32768) + } + return []Slot{slot(ops[rng.Intn(4)], reg(), base, off, 0)} + case 5: // store + ops := []uint8{OpStb, OpSth, OpStw, OpStdw, OpStxb, OpStxh, OpStxw, OpStxdw} + if ver == sbpfver.SbpfVersionV2 { + ops = []uint8{OpSt1BImm, OpSt2BImm, OpSt4BImm, OpSt8BImm, OpSt1BReg, OpSt2BReg, OpSt4BReg, OpSt8BReg} + } + var base uint8 + var off int16 + switch rng.Intn(5) { + case 0, 1: + base, off = 10, int16(-rng.Intn(4096)) + case 2: + base, off = 5, int16(rng.Intn(1024)) + case 3: + base, off = 1, int16(rng.Intn(600)) + default: + base, off = reg(), int16(rng.Intn(65536)-32768) + } + return []Slot{slot(ops[rng.Intn(8)], base, reg(), off, imm())} + case 6: // forward conditional jump (never backwards: guarantees termination) + maxOff := n - pc - 2 + if maxOff <= 0 { + return []Slot{slot(OpMov64Imm, reg(), 0, 0, imm())} + } + return []Slot{slot(jmp[rng.Intn(len(jmp))], reg(), reg(), int16(rng.Intn(min(maxOff, 8))), imm())} + case 7: // syscall + hs := []uint32{hashPoke, hashPeek, hashCopy, hashBurn, hashSetLen} + return []Slot{slot(OpCall, 0, 0, 0, hs[rng.Intn(len(hs))])} + case 8: // internal call + if ver >= sbpfver.SbpfVersionV3 { + return []Slot{slot(OpCall, 0, 1, 0, uint32(fnPC-int64(pc)-1))} + } + return []Slot{slot(OpCall, 0, 0, 0, PCHash(uint64(fnPC)))} + default: // set up pointer registers + switch rng.Intn(3) { + case 0: // r5 = heap + return diffLoadImm64(5, VaddrHeap, ver) + case 1: // r1 = input + small + return diffLoadImm64(1, VaddrInput+uint64(rng.Intn(64)), ver) + default: // r9 = random 64-bit + return diffLoadImm64(9, uint64(rng.Uint32())|uint64(rng.Intn(6))<<32, ver) + } + } +} + +func genProgram(rng *rand.Rand, ver uint32) *Program { + n := 8 + rng.Intn(120) + // layout: [0, n) main body then exit; fn at fnPC: a few ALU ops + exit + body := make([]Slot, 0, n+16) + fnPC := int64(n + 1) + for len(body) < n { + body = append(body, randSlot(rng, len(body), n, ver, fnPC)...) + } + if len(body) > n { + body = body[:n-1] // drop a cut lddw pair + } + body = append(body, slot(OpExit, 0, 0, 0, 0)) + // fix up forward jumps that would land on the second slot of an lddw + for pc := range body { + if body[pc].Op()&0x07 == ClassJmp && body[pc].Op() != OpCall && body[pc].Op() != OpExit { + dst := pc + int(body[pc].Off()) + 1 + if dst < len(body) && body[dst].Op() == 0 { + body[pc] = body[pc]&^(Slot(0xffff)<<16) | Slot(uint16(body[pc].Off()+1))<<16 + } + } + } + // function + store := uint8(OpStxdw) + if ver == sbpfver.SbpfVersionV2 { + store = OpSt8BReg + } + body = append(body, + slot(OpAdd64Imm, 6, 0, 0, uint32(rng.Intn(100))), + slot(OpXor64Reg, 7, 6, 0, 0), + slot(store, 10, 7, int16(-8-rng.Intn(64)), 0), + slot(OpExit, 0, 0, 0, 0)) + p := mkProgram(body, ver) + if ver < sbpfver.SbpfVersionV3 { + p.Funcs[PCHash(uint64(fnPC))] = fnPC + } + p.RO = make([]byte, 256) + for i := range p.RO { + p.RO[i] = byte(i * 7) + } + return p +} + +// Keep this check in the ordinary suite: merely selecting v2 is insufficient +// if its programs still contain legacy memory opcodes or LDDW and never run. +func TestDifferentialV2Generator(t *testing.T) { + rng := rand.New(rand.NewSource(12345)) + seen := make(map[uint8]bool) + for i := 0; i < 1000; i++ { + p := genProgram(rng, sbpfver.SbpfVersionV2) + if err := p.Verify(); err != nil { + continue + } + for _, ins := range p.Text { + seen[ins.Op()] = true + } + } + for _, op := range []uint8{OpLd1BReg, OpLd2BReg, OpLd4BReg, OpLd8BReg, + OpSt1BImm, OpSt2BImm, OpSt4BImm, OpSt8BImm, + OpSt1BReg, OpSt2BReg, OpSt4BReg, OpSt8BReg} { + if !seen[op] { + t.Errorf("no verifier-accepted v2 program contains opcode %#x", op) + } + } +} + +func memHash(bs ...[]byte) uint64 { + h := fnv.New64a() + for _, b := range bs { + h.Write(b) + } + return h.Sum64() +} + +func TestDifferentialDump(t *testing.T) { + out := os.Getenv("SBPF_DIFF_OUT") + if out == "" { + t.Skip("SBPF_DIFF_OUT not set") + } + f, err := os.Create(out) + if err != nil { + t.Fatal(err) + } + defer f.Close() + w := bufio.NewWriter(f) + defer w.Flush() + + writeDifferentialDump(t, w, 100000) +} + +func writeDifferentialDump(t *testing.T, w io.Writer, n int) { + rng := rand.New(rand.NewSource(12345)) + generated, verified := 0, 0 + var verifiedByVersion [4]int + for i := 0; i < n; i++ { + // Equal representation of every version, independent of RNG consumption. + ver := uint32(i % 4) + p := genProgram(rng, ver) + generated++ + if err := p.Verify(); err != nil { + fmt.Fprintf(w, "%d ver=%d VERIFY_FAIL %v\n", i, ver, err) + continue + } + verified++ + verifiedByVersion[ver]++ + resolveCallTargetsIfSupported(p) + input := make([]byte, 700) + for j := range input { + input[j] = byte(j) + } + var regions []InputRegion + useRegions := rng.Intn(2) == 0 + if useRegions { + regions = []InputRegion{ + {Offset: 0, HostOffset: 0, RegionSize: 100, AddressSpaceReserved: 100, Writable: true, AccountIndex: -1}, + {Offset: 100, HostOffset: 100, RegionSize: 150, AddressSpaceReserved: 300, Writable: rng.Intn(2) == 0, AccountIndex: 0}, + {Offset: 400, HostOffset: 400, RegionSize: 300, AddressSpaceReserved: 300, Writable: true, AccountIndex: -1}, + } + } + budget := uint64(1 + rng.Intn(400)) + if rng.Intn(4) == 0 { + budget = 100000 + } + cm := cu.NewComputeMeter(budget) + heapMax := 4096 * (1 + rng.Intn(4)) + ip := NewInterpreter(p, &VMOpts{ + HeapMax: heapMax, + Syscalls: diffRegistry, + ComputeMeter: &cm, + Input: input, + InputRegions: regions, + DisableStackFrameGaps: rng.Intn(3) == 0, + }) + var ret, cuUsed uint64 + var runErr error + func() { + defer func() { + if r := recover(); r != nil { + runErr = fmt.Errorf("PANIC: %v", r) + } + }() + ret, cuUsed, runErr = ip.Run() + }() + errStr := "" + if runErr != nil { + errStr = runErr.Error() + } + h := memHash(ip.stack.mem, ip.heap, input) + regionSizes := "" + for _, r := range ip.inputRegions { + regionSizes += fmt.Sprintf("%d/%v,", r.RegionSize, r.Writable) + } + fmt.Fprintf(w, "%d ver=%d budget=%d ret=%d cu=%d remaining=%d err=%q mem=%x regions=%s\n", + i, ver, budget, ret, cuUsed, cm.Remaining(), errStr, h, regionSizes) + ip.Finish() + if os.Getenv("SBPF_CHECK_POOL_ZERO") != "" { + // The buffers just returned to the pool must be all-zero. + st := stackMemPool.Get().([]byte) + hp := heapPool.Get().([]byte) + for j, b := range st[:StackMax] { + if b != 0 { + t.Fatalf("program %d: pooled stack not zeroed at %d", i, j) + } + } + for j, b := range hp[:cap(hp)] { + if b != 0 { + t.Fatalf("program %d: pooled heap not zeroed at %d", i, j) + } + } + stackMemPool.Put(st) + heapPool.Put(hp) + } + } + for ver, count := range verifiedByVersion { + if count == 0 { + t.Errorf("no verifier-accepted programs for v%d", ver) + } + } + t.Logf("generated=%d verified=%d verified_by_version=%v", generated, verified, verifiedByVersion) +} + +// Golden derived from the pre-optimization e1204b32 interpreter with this +// generator (seed 12345). Do not regenerate from candidate output alone. +// Covers 1024 programs per version; dumps compare results, errors, CU and memory. +func TestDifferentialGolden(t *testing.T) { + h := sha256.New() + writeDifferentialDump(t, h, 4096) + const want = "ccc16b4021e784342a7200a3d8911d362d2ab11a5ddefccf1c93880a704426b0" + if got := fmt.Sprintf("%x", h.Sum(nil)); got != want { + t.Fatalf("differential drift: got %s, want %s", got, want) + } +} diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go new file mode 100644 index 000000000..4a922a70f --- /dev/null +++ b/pkg/sbpf/pooling_test.go @@ -0,0 +1,135 @@ +package sbpf + +import ( + "bytes" + "encoding/binary" + "fmt" + "sync" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" + "github.com/stretchr/testify/require" +) + +func poolingInterpreter(heap int) *Interpreter { + meter := cu.NewComputeMeter(100) + return NewInterpreter(testV3Program([]Slot{testSlot(OpExit, 0, 0, 0, 0)}, nil), + &VMOpts{HeapMax: heap, ComputeMeter: &meter}) +} + +func TestPooledVMIsolationAcrossNestedAndConcurrentExecutions(t *testing.T) { + old := UsePool + UsePool = true + t.Cleanup(func() { UsePool = old }) + var wg sync.WaitGroup + for worker := 0; worker < 8; worker++ { + wg.Add(1) + go func(worker int) { + defer wg.Done() + for i := 0; i < 30; i++ { + parent := poolingInterpreter(32 * 1024) + if !bytes.Equal(parent.heap, make([]byte, len(parent.heap))) || + !bytes.Equal(parent.stack.mem, make([]byte, len(parent.stack.mem))) { + t.Error("pooled VM exposed data from an earlier execution") + } + // Write through the VM's translation layer, as programs and + // syscalls do: the pool only re-zeroes memory the VM saw written. + if err := parent.Write8(VaddrHeap, byte(worker+1)); err != nil { + t.Error(err) + } + if err := parent.Write8(VaddrStack, byte(worker+1)); err != nil { + t.Error(err) + } + child := poolingInterpreter(256 * 1024) + childHeap, err := child.Translate(VaddrHeap, uint64(len(child.heap)), true) + if err != nil { + t.Error(err) + } + for j := range childHeap { + childHeap[j] = 0xab + } + childStack, err := child.Translate(VaddrStack, StackMax, true) + if err != nil { + t.Error(err) + } + for j := range childStack { + childStack[j] = 0xcd + } + child.Finish() + if parent.heap[0] != byte(worker+1) || parent.stack.mem[0] != byte(worker+1) { + t.Error("nested VM storage aliased its active parent") + } + parent.Finish() + } + }(worker) + } + wg.Wait() + ip := poolingInterpreter(32 * 1024) + defer ip.Finish() + require.Len(t, ip.heap, 32*1024) + _, err := ip.Translate(VaddrHeap+32*1024, 1, false) + require.Error(t, err, "pool capacity must not widen the requested heap mapping") +} + +func BenchmarkVMCreateAndFinish(b *testing.B) { + old := UsePool + b.Cleanup(func() { UsePool = old }) + for _, pooled := range []bool{false, true} { + name := "fresh" + if pooled { + name = "pooled" + } + b.Run(name, func(b *testing.B) { + UsePool = pooled + b.ReportAllocs() + for i := 0; i < b.N; i++ { + ip := poolingInterpreter(32 * 1024) + ip.Finish() + } + }) + } +} + +// Exercise actual stores at and beyond the bitmap boundary. Inspect the returned +// buffer directly: sync.Pool is permitted to discard entries, so a subsequent Get +// alone would not reliably detect a missed clear. +func TestPooledHeapDirtyBitmapBoundary(t *testing.T) { + oldUsePool, oldPool := UsePool, heapPool + UsePool = true + heapPool = &sync.Pool{New: func() any { return newHeap() }} + t.Cleanup(func() { UsePool, heapPool = oldUsePool, oldPool }) + for _, size := range []int{fastDirtyBytes, fastDirtyBytes + 1, 2 * fastDirtyBytes} { + for _, ver := range []uint32{sbpfver.SbpfVersionV0, sbpfver.SbpfVersionV2, sbpfver.SbpfVersionV3} { + t.Run(fmt.Sprintf("%d/v%d", size, ver), func(t *testing.T) { + offsets := []int{0, size - 8} + if size >= fastDirtyBytes+8 { + offsets = append(offsets, fastDirtyBytes-4, fastDirtyBytes) + } + var text []Slot + op := uint8(OpStdw) + if ver == sbpfver.SbpfVersionV2 { + op = OpSt8BImm + } + for _, off := range offsets { + text = append(text, diffLoadImm64(5, VaddrHeap+uint64(off), ver)...) + text = append(text, slot(op, 5, 0, 0, 0x12345678)) + } + text = append(text, slot(OpExit, 0, 0, 0, 0)) + program := mkProgram(text, ver) + require.NoError(t, program.Verify()) + meter := cu.NewComputeMeter(100) + ip := NewInterpreter(program, &VMOpts{HeapMax: size, ComputeMeter: &meter, Syscalls: noSyscalls}) + // Always clear test storage on failure so later cases cannot inherit dirt. + defer func() { clear(ip.heap) }() + require.NotNil(t, ip.fastRead(VaddrHeap+uint64(size-8), 8)) + _, _, err := ip.Run() + require.NoError(t, err) + require.Equal(t, uint64(0x12345678), binary.LittleEndian.Uint64(ip.heap[offsets[len(offsets)-1]:])) + ip.Finish() + require.True(t, bytes.Equal(make([]byte, size), ip.heap), "Finish must clear every written byte before pooling") + require.Equal(t, size <= fastDirtyBytes, ip.regions[VaddrHeap>>32].wlen != 0) + }) + } + } +} diff --git a/pkg/sbpf/program.go b/pkg/sbpf/program.go index c112be14d..a353a4594 100644 --- a/pkg/sbpf/program.go +++ b/pkg/sbpf/program.go @@ -13,13 +13,36 @@ type Program struct { Entrypoint uint64 // PC Funcs map[uint32]int64 SbpfVersion sbpfver.SbpfVersion + + // CallTargets[pc] holds the resolved internal function target for a + // `call imm` slot at pc (non-static-syscall versions), or -1. + CallTargets []int64 +} + +// ResolveCallTargets precomputes CallTargets from Funcs so the interpreter +// does not need a map lookup per call instruction. +func (p *Program) ResolveCallTargets() { + if p.SbpfVersion.EnableStaticSyscalls() { + p.CallTargets = nil + return + } + targets := make([]int64, len(p.Text)) + for pc, slot := range p.Text { + targets[pc] = -1 + if slot.Op() == OpCall { + if t, ok := p.Funcs[slot.Uimm()]; ok { + targets[pc] = t + } + } + } + p.CallTargets = targets } func (p *Program) MemoryBytes() uint64 { if p == nil { return 0 } - total := uint64(len(p.RO)) + uint64(len(p.Text))*8 + total := uint64(len(p.RO)) + uint64(len(p.Text))*8 + uint64(len(p.CallTargets))*8 if len(p.RO) == 0 { total += uint64(len(p.TextBytes)) } diff --git a/pkg/sbpf/program_memory_test.go b/pkg/sbpf/program_memory_test.go new file mode 100644 index 000000000..3360cf45d --- /dev/null +++ b/pkg/sbpf/program_memory_test.go @@ -0,0 +1,12 @@ +package sbpf + +import "testing" + +func TestProgramMemoryIncludesResolvedCalls(t *testing.T) { + p := &Program{Text: make([]Slot, 20)} + before := p.MemoryBytes() + p.ResolveCallTargets() + if got := p.MemoryBytes() - before; got != 20*8 { + t.Fatalf("resolved-call cache bytes = %d, want 160", got) + } +} diff --git a/pkg/sbpf/stack.go b/pkg/sbpf/stack.go index 3f3331b6e..edc662c18 100644 --- a/pkg/sbpf/stack.go +++ b/pkg/sbpf/stack.go @@ -36,6 +36,12 @@ type Stack struct { shadow []Frame dynamicStackFrames bool stackFrameGaps bool + // dirtyLo/dirtyHi bound the physical byte range of mem that may have been + // written during this execution. Finish only has to zero this range before + // returning the buffer to the pool. + dirtyLo uint64 + dirtyHi uint64 + maxDepth int } // Frame is an entry on the shadow stack. @@ -92,9 +98,11 @@ func NewStack(sbpfVer sbpfver.SbpfVersion, disableStackFrameGaps bool) Stack { } s := Stack{ - mem: m, - sp: VaddrStack, - shadow: sh, + mem: m, + sp: VaddrStack, + shadow: sh, + dirtyLo: StackMax, + dirtyHi: 0, } var sz uint64 @@ -115,10 +123,12 @@ func NewStack(sbpfVer sbpfver.SbpfVersion, disableStackFrameGaps bool) Stack { func (s *Stack) Finish() { if UsePool { s.mem = s.mem[:StackMax] - clear(s.mem) + if s.dirtyHi > s.dirtyLo { + clear(s.mem[s.dirtyLo:s.dirtyHi]) + } stackMemPool.Put(s.mem) s.shadow = s.shadow[:StackDepth] - clear(s.shadow) + clear(s.shadow[:max(s.maxDepth, 1)]) s.shadow = s.shadow[:1] stackShadowPool.Put(s.shadow) } @@ -153,21 +163,38 @@ func (s *Stack) GetFrame(addr uint32) []byte { } } +// MarkDirty records that physical stack bytes [off, off+size) may be written. +func (s *Stack) MarkDirty(lo, hi uint64) { + if hi <= lo { + return + } + s.dirtyLo = min(s.dirtyLo, lo) + s.dirtyHi = max(s.dirtyHi, hi) +} + // Push allocates a new call frame. // // Saves the given nonvolatile regs, return address, // and current frame pointer. // Returns the new frame pointer. -func (s *Stack) Push(regs []uint64, ret int64) bool { - if ok := len(s.shadow) < cap(s.shadow); !ok { +func (s *Stack) Push(regs *[16]uint64, ret int64) bool { + n := len(s.shadow) + if n >= cap(s.shadow) { return false } - - frame := Frame{RetAddr: ret} - copy(frame.NVRegs[:], regs[6:10]) - frame.FramePtr = regs[10] - - s.shadow = append(s.shadow, frame) + // Write the frame in place (no temporary Frame value / copy) to avoid + // store-forwarding stalls in this very hot path. + s.shadow = s.shadow[:n+1] + f := &s.shadow[n] + f.RetAddr = ret + f.NVRegs[0] = regs[6] + f.NVRegs[1] = regs[7] + f.NVRegs[2] = regs[8] + f.NVRegs[3] = regs[9] + f.FramePtr = regs[10] + if n+1 > s.maxDepth { + s.maxDepth = n + 1 + } if !s.dynamicStackFrames { if s.stackFrameGaps { @@ -185,16 +212,17 @@ func (s *Stack) Push(regs []uint64, ret int64) bool { // Restores saved nonvolatile regs into provided slice. // Returns saved return address and returns true upon success, // and returns false if no call frames are left. -func (s *Stack) Pop(regs []uint64) (int64, bool) { - if len(s.shadow) <= 1 { +func (s *Stack) Pop(regs *[16]uint64) (int64, bool) { + n := len(s.shadow) + if n <= 1 { return 0, false } - - var frame Frame - frame, s.shadow = s.shadow[len(s.shadow)-1], s.shadow[:len(s.shadow)-1] - - copy(regs[6:10], frame.NVRegs[:]) - regs[10] = frame.FramePtr - - return frame.RetAddr, true + f := &s.shadow[n-1] + regs[6] = f.NVRegs[0] + regs[7] = f.NVRegs[1] + regs[8] = f.NVRegs[2] + regs[9] = f.NVRegs[3] + regs[10] = f.FramePtr + s.shadow = s.shadow[:n-1] + return f.RetAddr, true } diff --git a/pkg/sbpf/translate_overflow_test.go b/pkg/sbpf/translate_overflow_test.go new file mode 100644 index 000000000..98d8ee2d0 --- /dev/null +++ b/pkg/sbpf/translate_overflow_test.go @@ -0,0 +1,25 @@ +package sbpf + +import ( + "github.com/stretchr/testify/require" + "testing" +) + +func TestTranslateRejectsOverflowingRange(t *testing.T) { + ip := &Interpreter{ro: make([]byte, 32), heap: make([]byte, 32), input: make([]byte, 32)} + for _, base := range []uint64{VaddrProgram, VaddrHeap, VaddrInput} { + for _, write := range []bool{false, true} { + for _, offset := range []uint64{1, 31, 33, 0xffffffff} { + for _, size := range []uint64{^uint64(0), ^uint64(0) - 15} { + _, err := ip.Translate(base+offset, size, write) + require.Error(t, err, "base=%x offset=%d size=%d write=%v", base, offset, size, write) + } + } + } + got, err := ip.Translate(base+31, 1, false) + require.NoError(t, err) + require.Len(t, got, 1) + _, err = ip.Translate(base+31, 2, false) + require.Error(t, err) + } +} diff --git a/pkg/sbpf/vasa_test.go b/pkg/sbpf/vasa_test.go index e7d64034c..77912c07a 100644 --- a/pkg/sbpf/vasa_test.go +++ b/pkg/sbpf/vasa_test.go @@ -73,7 +73,7 @@ func TestStackFrameGapsCanBeDisabled(t *testing.T) { gapped := NewStack(version, false) defer gapped.Finish() - gappedRegs := make([]uint64, 11) + gappedRegs := new([16]uint64) gappedRegs[10] = VaddrStack + StackFrameSize require.True(t, gapped.Push(gappedRegs, 0)) require.Equal(t, VaddrStack+StackFrameSize*3, gappedRegs[10]) @@ -81,7 +81,7 @@ func TestStackFrameGapsCanBeDisabled(t *testing.T) { contiguous := NewStack(version, true) defer contiguous.Finish() - contiguousRegs := make([]uint64, 11) + contiguousRegs := new([16]uint64) contiguousRegs[10] = VaddrStack + StackFrameSize require.True(t, contiguous.Push(contiguousRegs, 0)) require.Equal(t, VaddrStack+StackFrameSize*2, contiguousRegs[10]) @@ -94,9 +94,22 @@ func TestStackFrameGapsAreLegacyOnly(t *testing.T) { stack := NewStack(version, false) defer stack.Finish() - regs := make([]uint64, 11) + regs := new([16]uint64) regs[10] = VaddrStack + StackFrameSize require.True(t, stack.Push(regs, 0)) require.Equal(t, VaddrStack+StackFrameSize*2, regs[10]) require.NotNil(t, stack.GetFrame(StackFrameSize)) } + +func TestInputRegionFastCacheClampsToBackingBytes(t *testing.T) { + ip := &Interpreter{input: make([]byte, 8), inputRegions: []InputRegion{{Offset: 0, HostOffset: 4, RegionSize: 100, AddressSpaceReserved: 100, Writable: true, AccountIndex: -1}}} + ip.initRegions() + require.NoError(t, ip.Write8(VaddrInput, 7)) + require.NotNil(t, ip.fastRead(VaddrInput+3, 1)) + require.Nil(t, ip.fastRead(VaddrInput+4, 1)) + require.Nil(t, ip.fastWrite(VaddrInput+4, 1)) + _, err := ip.Read8(VaddrInput + 4) + require.Error(t, err) + require.Error(t, ip.Write8(VaddrInput+4, 8)) + require.Equal(t, byte(7), ip.input[4]) +} diff --git a/pkg/sealevel/bpf_loader.go b/pkg/sealevel/bpf_loader.go index b80ab8853..16ba486cc 100644 --- a/pkg/sealevel/bpf_loader.go +++ b/pkg/sealevel/bpf_loader.go @@ -123,7 +123,11 @@ func (write *UpgradeableLoaderInstrWrite) MarshalWithEncoder(encoder *bin.Encode return err } - err = encoder.WriteBytes(write.Bytes, true) + // UpgradeableLoaderInstruction uses bincode's fixed-width u64 vector length. + if err = encoder.WriteUint64(uint64(len(write.Bytes)), bin.LE); err != nil { + return err + } + err = encoder.WriteBytes(write.Bytes, false) return err } @@ -1338,6 +1342,11 @@ func executeLoadedProgram(execCtx *ExecutionCtx, program *sbpf.Program, syscallR func executeProgramFromBytes(execCtx *ExecutionCtx, programAddr solana.PublicKey, programData []byte, syscallRegistry sbpf.SyscallRegistry) error { start := time.Now() + // The caller has already validated the bank-visible loader metadata. Never + // let a global cache hit bypass that validation or select a different fork. + if entry, ok := execCtx.SlotCtx.AccountsDb.MaybeGetProgramFromCache(programAddr); ok && entry.MatchesSource(programData, &execCtx.Features) { + return executeLoadedProgram(execCtx, entry.Program, syscallRegistry) + } loader, err := loader.NewLoaderWithSyscalls(programData, syscallRegistry, false, &execCtx.Features) if err != nil { return InstrErrUnsupportedProgramId @@ -1352,6 +1361,7 @@ func executeProgramFromBytes(execCtx *ExecutionCtx, programAddr solana.PublicKey } entry := &accountsdb.ProgramCacheEntry{Program: program} + entry.BindSource(programData, &execCtx.Features) if !execCtx.IsSimulation { addProgramToCache(execCtx, programAddr, entry) } @@ -1361,11 +1371,16 @@ func executeProgramFromBytes(execCtx *ExecutionCtx, programAddr solana.PublicKey return executeLoadedProgram(execCtx, program, syscallRegistry) } +// All loader cache insertions must pass through this helper. A speculative +// stream records replacements for eviction on discard; the previous program is +// then reloaded from authoritative account data, never retained from that stream. +// Replay must discard its stream before executing a different bank. func addProgramToCache(execCtx *ExecutionCtx, programAddr solana.PublicKey, entry *accountsdb.ProgramCacheEntry) { if execCtx.SlotCtx == nil || execCtx.SlotCtx.AccountsDb == nil { return } execCtx.SlotCtx.AccountsDb.AddProgramToCache(programAddr, entry) + execCtx.SlotCtx.RecordProgramCacheAdd(programAddr) } func mapVirtualAddressSpaceRunErr(execCtx *ExecutionCtx, err error, inputRegions []sbpf.InputRegion) error { @@ -1486,36 +1501,27 @@ func BpfLoaderProgramExecute(execCtx *ExecutionCtx) error { } var programBytes []byte - var loadedProgram *sbpf.Program - var hasLoadedProgram bool var programAcctKey solana.PublicKey programOwner := programAcct.Owner() if programOwner == a.BpfLoader2Addr || programOwner == a.BpfLoaderDeprecatedAddr { - var programCacheEntry *accountsdb.ProgramCacheEntry - programCacheEntry, hasLoadedProgram = execCtx.SlotCtx.AccountsDb.MaybeGetProgramFromCache(programAcct.Key()) - if hasLoadedProgram { - programAcctKey = programAcct.Key() - loadedProgram = programCacheEntry.Program - } else { // program is not cached - if len(programAcct.Data()) == 0 { - var paTmp *accounts.Account - paTmp, err = execCtx.SlotCtx.GetAccount(programAcct.Key()) + if len(programAcct.Data()) == 0 { + var paTmp *accounts.Account + paTmp, err = execCtx.SlotCtx.GetAccount(programAcct.Key()) + if err != nil { + paTmp, err = execCtx.SlotCtx.GetAccountFromAccountsDb(programAcct.Key()) if err != nil { - paTmp, err = execCtx.SlotCtx.GetAccountFromAccountsDb(programAcct.Key()) - if err != nil { - //mlog.Log.Debugf("unable to get account %s from accountsdb", programAcct.Key()) - return InstrErrUnsupportedProgramId - } + //mlog.Log.Debugf("unable to get account %s from accountsdb", programAcct.Key()) + return InstrErrUnsupportedProgramId } - programBytes = paTmp.Data - } else { - programBytes = programAcct.Data() } - programAcctKey = programAcct.Key() + programBytes = paTmp.Data + } else { + programBytes = programAcct.Data() } + programAcctKey = programAcct.Key() } else if programOwner == a.BpfLoaderUpgradeableAddr { var programAcctState *UpgradeableLoaderState @@ -1543,49 +1549,38 @@ func BpfLoaderProgramExecute(execCtx *ExecutionCtx) error { } start := time.Now() - var programCacheEntry *accountsdb.ProgramCacheEntry - programCacheEntry, hasLoadedProgram = execCtx.SlotCtx.AccountsDb.MaybeGetProgramFromCache(programAcctState.Program.ProgramDataAddress) - if hasLoadedProgram { - if programCacheEntry.DeploymentSlot >= execCtx.SlotCtx.Slot { - return InstrErrInvalidAccountData - } - programAcctKey = programAcctState.Program.ProgramDataAddress - loadedProgram = programCacheEntry.Program - metrics.GlobalBlockReplay.GetProgramDataCached.AddTimingSince(start) - } else { // program is not cached - programDataAcct, err := execCtx.SlotCtx.GetAccount(programAcctState.Program.ProgramDataAddress) + programDataAcct, err := execCtx.SlotCtx.GetAccount(programAcctState.Program.ProgramDataAddress) + if err != nil { + programDataAcct, err = execCtx.SlotCtx.GetAccountFromAccountsDb(programAcctState.Program.ProgramDataAddress) if err != nil { - programDataAcct, err = execCtx.SlotCtx.GetAccountFromAccountsDb(programAcctState.Program.ProgramDataAddress) - if err != nil { - return InstrErrUnsupportedProgramId - } - metrics.GlobalBlockReplay.GetProgramDataUncachedAccountsDb.AddTimingSince(start) - } else { - metrics.GlobalBlockReplay.GetProgramDataUncachedAccounts.AddTimingSince(start) + return InstrErrUnsupportedProgramId } + metrics.GlobalBlockReplay.GetProgramDataUncachedAccountsDb.AddTimingSince(start) + } else { + metrics.GlobalBlockReplay.GetProgramDataUncachedAccounts.AddTimingSince(start) + } - start = time.Now() - programDataAcctState, err := UnmarshalUpgradeableLoaderState(programDataAcct.Data) - if err != nil { - return err - } + start = time.Now() + programDataAcctState, err := UnmarshalUpgradeableLoaderState(programDataAcct.Data) + if err != nil { + return err + } - if programDataAcctState.Type != UpgradeableLoaderStateTypeProgramData { - return InstrErrUnsupportedProgramId - } + if programDataAcctState.Type != UpgradeableLoaderStateTypeProgramData { + return InstrErrUnsupportedProgramId + } - programDataSlot := programDataAcctState.ProgramData.Slot - if programDataSlot >= execCtx.SlotCtx.Slot { - return InstrErrInvalidAccountData - } + programDataSlot := programDataAcctState.ProgramData.Slot + if programDataSlot >= execCtx.SlotCtx.Slot { + return InstrErrInvalidAccountData + } - if len(programDataAcct.Data) < upgradeableLoaderSizeOfProgramDataMetaData { - return InstrErrUnsupportedProgramId - } - programAcctKey = programAcctState.Program.ProgramDataAddress - programBytes = programDataAcct.Data[upgradeableLoaderSizeOfProgramDataMetaData:] - metrics.GlobalBlockReplay.GetProgramDataUncachedMarshal.AddTimingSince(start) + if len(programDataAcct.Data) < upgradeableLoaderSizeOfProgramDataMetaData { + return InstrErrUnsupportedProgramId } + programAcctKey = programAcctState.Program.ProgramDataAddress + programBytes = programDataAcct.Data[upgradeableLoaderSizeOfProgramDataMetaData:] + metrics.GlobalBlockReplay.GetProgramDataUncachedMarshal.AddTimingSince(start) } else { return InstrErrUnsupportedProgramId } @@ -1596,13 +1591,7 @@ func BpfLoaderProgramExecute(execCtx *ExecutionCtx) error { return Syscalls(&execCtx.Features, false, u) }) - // two cases here: we're either executing from the program cache, so from a pre-parsed/loaded program, or from bytes if - // the the program was not found in the cache. - if hasLoadedProgram { - err = executeLoadedProgram(execCtx, loadedProgram, syscallRegistry) - } else { - err = executeProgramFromBytes(execCtx, programAcctKey, programBytes, syscallRegistry) - } + err = executeProgramFromBytes(execCtx, programAcctKey, programBytes, syscallRegistry) return err } diff --git a/pkg/sealevel/execution_ctx.go b/pkg/sealevel/execution_ctx.go index 348b4647c..326da6762 100644 --- a/pkg/sealevel/execution_ctx.go +++ b/pkg/sealevel/execution_ctx.go @@ -5,7 +5,6 @@ import ( "fmt" "sync" "sync/atomic" - "time" "github.com/Overclock-Validator/mithril/pkg/accounts" "github.com/Overclock-Validator/mithril/pkg/accountsdb" @@ -19,6 +18,9 @@ import ( ) type ExecutionCtx struct { + // SkipTimingMetrics disables instruction-dispatch timing collection for + // leader execution; it never changes instruction validation or CU charging. + SkipTimingMetrics bool Log Logger Accounts accounts.Accounts TransactionContext *TransactionCtx @@ -105,6 +107,50 @@ type SlotCtx struct { bankSysvars atomic.Pointer[BankSysvars] TraceCtx context.Context + + // Speculative execution support (streaming replay). A bank executed before + // its block is complete must not publish to process-global state until + // the block is accepted, so a discard leaves nothing behind. + // + // DeferVoteCachePublication buffers global vote-cache puts and deletes in + // the pending maps (protected by PendingVoteCacheMu) and turns the + // vote/stake dirty marker into VoteStakeDirty; the replay finalize step + // publishes them once the bank is accepted. + DeferVoteCachePublication bool + PendingVoteCacheMu sync.Mutex + PendingVoteCache map[solana.PublicKey]*VoteStateVersions + PendingVoteCacheDeletes map[solana.PublicKey]struct{} + VoteStakeDirty bool + // TrackProgramCacheAdds records every program-cache insertion made while + // executing this bank so a discarded bank can evict them again (eviction + // only forces a reload from account data, so it is always safe). + TrackProgramCacheAdds bool + ProgramCacheAddsMu sync.Mutex + ProgramCacheAdds []solana.PublicKey +} + +// RecordProgramCacheAdd notes a program-cache insertion for later undo when +// TrackProgramCacheAdds is set. +func (slotCtx *SlotCtx) RecordProgramCacheAdd(key solana.PublicKey) { + if slotCtx == nil || !slotCtx.TrackProgramCacheAdds { + return + } + slotCtx.ProgramCacheAddsMu.Lock() + slotCtx.ProgramCacheAdds = append(slotCtx.ProgramCacheAdds, key) + slotCtx.ProgramCacheAddsMu.Unlock() +} + +// TakeProgramCacheAdds returns and clears the recorded program-cache +// insertions. +func (slotCtx *SlotCtx) TakeProgramCacheAdds() []solana.PublicKey { + if slotCtx == nil { + return nil + } + slotCtx.ProgramCacheAddsMu.Lock() + defer slotCtx.ProgramCacheAddsMu.Unlock() + adds := slotCtx.ProgramCacheAdds + slotCtx.ProgramCacheAdds = nil + return adds } // BankSysvars returns the immutable sysvar snapshot owned by this bank. @@ -237,14 +283,14 @@ func (execCtx *ExecutionCtx) PrepareInstruction(ix Instruction, signers []solana } func (execCtx *ExecutionCtx) ProcessInstruction(instrData []byte, instructionAccts []InstructionAccount, programIndices []uint64) error { - start := time.Now() + start := metrics.StartTiming(!execCtx.SkipTimingMetrics) nextInstrCtx, err := execCtx.TransactionContext.NextInstructionCtx() if err != nil { return err } metrics.GlobalBlockReplay.GetNextIxCtx.AddTimingSince(start) - start = time.Now() + start = metrics.StartTiming(!execCtx.SkipTimingMetrics) nextInstrCtx.Configure(programIndices, instructionAccts, instrData) metrics.GlobalBlockReplay.NextIxCtxConfigure.AddTimingSince(start) @@ -267,7 +313,7 @@ func (execCtx *ExecutionCtx) ProcessInstruction(instrData []byte, instructionAcc }) } - start = time.Now() + start = metrics.StartTiming(!execCtx.SkipTimingMetrics) err = execCtx.Push() if err != nil { return err @@ -283,7 +329,7 @@ func (execCtx *ExecutionCtx) ProcessInstruction(instrData []byte, instructionAcc err1 := execCtx.ExecuteInstruction() - start = time.Now() + start = metrics.StartTiming(!execCtx.SkipTimingMetrics) err2 := execCtx.Pop() metrics.GlobalBlockReplay.IxPop.AddTimingSince(start) @@ -304,7 +350,7 @@ func (execCtx *ExecutionCtx) AddModifiedVoteState(pubkey solana.PublicKey, state } func (execCtx *ExecutionCtx) ExecuteInstruction() error { - start := time.Now() + start := metrics.StartTiming(!execCtx.SkipTimingMetrics) txCtx := execCtx.TransactionContext instrCtx, err := txCtx.CurrentInstructionCtx() @@ -334,7 +380,7 @@ func (execCtx *ExecutionCtx) ExecuteInstruction() error { } metrics.GlobalBlockReplay.ExecIxResolveNativeProgram.AddTimingSince(start) - start = time.Now() + start = metrics.StartTiming(!execCtx.SkipTimingMetrics) err = nativeProgramFn(execCtx) switch nativeProgramStr { case a.SystemProgramAddrStr: diff --git a/pkg/sealevel/legacy_bank_fixture_test.go b/pkg/sealevel/legacy_bank_fixture_test.go new file mode 100644 index 000000000..cbf7cf81a --- /dev/null +++ b/pkg/sealevel/legacy_bank_fixture_test.go @@ -0,0 +1,37 @@ +package sealevel + +import ( + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/gagliardetto/solana-go" + "github.com/maypok86/otter" + "github.com/stretchr/testify/require" + "testing" +) + +// Legacy instruction fixtures predate the bank and program-cache dependencies +// used by loaded programs. Give each fixture isolated state, as replay does. +func initializeLegacyBankFixture(t *testing.T, ctx *ExecutionCtx) { + t.Helper() + ctx.RecordInnerInstructions = true + if ctx.Log == nil { + ctx.Log = &LogRecorder{} + } + t.Cleanup(func() { + if t.Failed() { + t.Logf("program logs: %#v", ctx.Log) + } + }) + cache, err := otter.MustBuilder[solana.PublicKey, *accountsdb.ProgramCacheEntry](1024).Cost(func(solana.PublicKey, *accountsdb.ProgramCacheEntry) uint32 { return 1 }).Build() + require.NoError(t, err) + t.Cleanup(cache.Close) + if ctx.Accounts == nil { + ctx.Accounts = accounts.NewMemAccounts() + } + if ctx.SlotCtx == nil { + ctx.SlotCtx = &SlotCtx{} + } + ctx.SlotCtx.Accounts = ctx.Accounts + ctx.SlotCtx.AccountsDb = &accountsdb.AccountsDb{ProgramCache: cache} + ctx.TransactionContext.ComputeBudgetLimits = &ComputeBudgetLimits{UpdatedHeapBytes: 32768} +} diff --git a/pkg/sealevel/loader_v4.go b/pkg/sealevel/loader_v4.go index 6bb264b21..8ab33effb 100644 --- a/pkg/sealevel/loader_v4.go +++ b/pkg/sealevel/loader_v4.go @@ -274,38 +274,28 @@ func LoaderV4Execute(execCtx *ExecutionCtx) error { return err } - var loadedProgram *sbpf.Program var programBytes []byte - - programCacheEntry, hasLoadedProgram := execCtx.SlotCtx.AccountsDb.MaybeGetProgramFromCache(program.Key()) - if hasLoadedProgram { - if programCacheEntry.DeploymentSlot >= execCtx.SlotCtx.Slot { - return InstrErrInvalidAccountData - } - loadedProgram = programCacheEntry.Program - } else { - programDataAcct, err := execCtx.SlotCtx.GetAccount(program.Key()) - if err != nil { - programDataAcct, err = execCtx.SlotCtx.GetAccountFromAccountsDb(program.Key()) - if err != nil { - return InstrErrUnsupportedProgramId - } - } - - state, err := decodeLoaderV4State(programDataAcct.Data) + programDataAcct, err := execCtx.SlotCtx.GetAccount(program.Key()) + if err != nil { + programDataAcct, err = execCtx.SlotCtx.GetAccountFromAccountsDb(program.Key()) if err != nil { return InstrErrUnsupportedProgramId } + } - if state.Status == LoaderV4StatusRetracted { - return InstrErrUnsupportedProgramId - } - if state.Slot >= execCtx.SlotCtx.Slot { - return InstrErrUnsupportedProgramId - } + state, err := decodeLoaderV4State(programDataAcct.Data) + if err != nil { + return InstrErrUnsupportedProgramId + } - programBytes = programDataAcct.Data[loaderV4ProgramDataOffset:] + if state.Status == LoaderV4StatusRetracted { + return InstrErrUnsupportedProgramId } + if state.Slot >= execCtx.SlotCtx.Slot { + return InstrErrUnsupportedProgramId + } + + programBytes = programDataAcct.Data[loaderV4ProgramDataOffset:] syscallRegistry := sbpf.SyscallRegistry(func(u uint32) (sbpf.Syscall, bool) { return Syscalls(&execCtx.Features, false, u) @@ -313,13 +303,8 @@ func LoaderV4Execute(execCtx *ExecutionCtx) error { program.Drop() - // two cases here: we're either executing from the program cache, so from a pre-parsed/loaded program, or from bytes if - // the the program was not found in the cache. - if hasLoadedProgram { - err = executeLoadedProgram(execCtx, loadedProgram, syscallRegistry) - } else { - err = executeProgramFromBytes(execCtx, program.Key(), programBytes, syscallRegistry) - } + err = executeProgramFromBytes(execCtx, program.Key(), programBytes, syscallRegistry) + } return err @@ -627,6 +612,7 @@ func LoaderV4ProcessDeploy(execCtx *ExecutionCtx) error { entry := &accountsdb.ProgramCacheEntry{Program: programObj, DeploymentSlot: currentSlot} if !execCtx.IsSimulation { execCtx.SlotCtx.AccountsDb.AddProgramToCache(program.Key(), entry) + execCtx.SlotCtx.RecordProgramCacheAdd(program.Key()) } return nil diff --git a/pkg/sealevel/loader_wire_test.go b/pkg/sealevel/loader_wire_test.go new file mode 100644 index 000000000..e45d375e8 --- /dev/null +++ b/pkg/sealevel/loader_wire_test.go @@ -0,0 +1,30 @@ +package sealevel + +import ( + "bytes" + "encoding/binary" + bin "github.com/gagliardetto/binary" + "github.com/stretchr/testify/require" + "testing" +) + +func TestUpgradeableLoaderWriteWireLayout(t *testing.T) { + instruction := UpgradeableLoaderInstrWrite{Offset: 0x12345678, Bytes: []byte{0xAB, 0xCD}} + var buf bytes.Buffer + require.NoError(t, instruction.MarshalWithEncoder(bin.NewBinEncoder(&buf))) + require.Equal(t, []byte{1, 0, 0, 0, 0x78, 0x56, 0x34, 0x12, 2, 0, 0, 0, 0, 0, 0, 0, 0xAB, 0xCD}, buf.Bytes()) + var decoded UpgradeableLoaderInstrWrite + require.NoError(t, decoded.UnmarshalWithDecoder(bin.NewBinDecoder(buf.Bytes()[4:]))) + require.Equal(t, instruction, decoded) +} + +func TestSolAccountMetaCWireLayout(t *testing.T) { + for _, bits := range [][2]byte{{0, 0}, {1, 0}, {0, 1}, {1, 1}} { + meta := SolAccountMetaC{PubkeyAddr: 0x12345678, IsWritable: bits[0], IsSigner: bits[1]} + wire, err := meta.Marshal() + require.NoError(t, err) + want := binary.LittleEndian.AppendUint64(nil, meta.PubkeyAddr) + want = append(want, bits[:]...) + require.Equal(t, want, wire) + } +} diff --git a/pkg/sealevel/native_perf_bench_test.go b/pkg/sealevel/native_perf_bench_test.go new file mode 100644 index 000000000..447931695 --- /dev/null +++ b/pkg/sealevel/native_perf_bench_test.go @@ -0,0 +1,287 @@ +package sealevel + +import ( + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/gagliardetto/solana-go" +) + +// Micro-benchmarks for the natively implemented programs that are still on the +// mainnet hot path (System transfer, Vote TowerSync), measured through the same +// ExecutionCtx.ProcessInstruction entry point replay uses, so the per-instruction +// plumbing (instruction context push/pop, lamport-sum checks, timing metrics) +// is included. Run with: +// +// go test ./pkg/sealevel/ -run XXX -bench 'Native|VoteState|Timing' -benchmem -cpu 1 -count 5 + +func benchPubkey(b byte) solana.PublicKey { + var pk solana.PublicKey + for i := range pk { + pk[i] = b + } + return pk +} + +// newBenchExecCtx mirrors newSystemProgramTestExecCtx without testing.T. +func newBenchExecCtx(txAccts *TransactionAccounts, clockSlot uint64, enabled ...features.FeatureGate) *ExecutionCtx { + txCtx := NewTransactionCtx(*txAccts, 5, 64) + execCtx := &ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeter(1 << 62)} + execCtx.Accounts = accounts.NewMemAccounts() + + clockAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct); err != nil { + panic(err) + } + WriteClockSysvar(&execCtx.Accounts, SysvarClock{Slot: clockSlot, Epoch: 0}) + + rentAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct); err != nil { + panic(err) + } + WriteRentSysvar(&execCtx.Accounts, SysvarRent{LamportsPerUint8Year: 3480, ExemptionThreshold: 2, BurnPercent: 50}) + + f := features.NewFeaturesDefault() + for _, gate := range enabled { + f.EnableFeature(gate, 0) + } + execCtx.Features = *f + return execCtx +} + +// resetTxCtx gives the execution context a fresh instruction trace/stack for +// the next instruction (each ProcessInstruction consumes one trace slot). +func resetTxCtx(execCtx *ExecutionCtx, txAccts *TransactionAccounts) { + execCtx.TransactionContext = NewTransactionCtx(*txAccts, 5, 64) +} + +// ---------------------------------------------------------------- System + +func encodeSystemTransfer(lamports uint64) []byte { + out := binary.LittleEndian.AppendUint32(nil, uint32(SystemProgramInstrTypeTransfer)) + return binary.LittleEndian.AppendUint64(out, lamports) +} + +func benchmarkSystemTransfer(b *testing.B, skipTiming bool) { + systemProgramAcct := accounts.Account{Key: a.SystemProgramAddr, Lamports: 1, Data: []byte{}, Owner: a.NativeLoaderAddr, Executable: true} + from := accounts.Account{Key: benchPubkey(0x11), Lamports: 1 << 60, Data: []byte{}, Owner: a.SystemProgramAddr} + to := accounts.Account{Key: benchPubkey(0x22), Lamports: 1_000_000, Data: []byte{}, Owner: a.SystemProgramAddr} + txAccts := NewTransactionAccounts([]accounts.Account{systemProgramAcct, from, to}) + metas := []AccountMeta{ + {Pubkey: from.Key, IsSigner: true, IsWritable: true}, + {Pubkey: to.Key, IsSigner: false, IsWritable: true}, + } + instrAccts := InstructionAcctsFromAccountMetas(metas, *txAccts) + instr := encodeSystemTransfer(1) + + execCtx := newBenchExecCtx(txAccts, 1234) + execCtx.SkipTimingMetrics = skipTiming + + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + resetTxCtx(execCtx, txAccts) + if err := execCtx.ProcessInstruction(instr, instrAccts, []uint64{0}); err != nil { + b.Fatal(err) + } + } + b.StopTimer() + if got := txAccts.Accounts[2].Lamports; got != 1_000_000+uint64(b.N) { + b.Fatalf("destination lamports %d, want %d", got, 1_000_000+uint64(b.N)) + } +} + +func BenchmarkNativeSystemTransfer(b *testing.B) { benchmarkSystemTransfer(b, false) } +func BenchmarkNativeSystemTransferNoTiming(b *testing.B) { benchmarkSystemTransfer(b, true) } + +// ---------------------------------------------------------------- Vote + +const ( + benchVoteRoot = uint64(1000) + benchVoteLastSlot = benchVoteRoot + MaxLockoutHistory // 1031 + benchClockSlot = benchVoteLastSlot + 2 +) + +func benchSlotHash(slot uint64) [32]byte { + var h [32]byte + binary.LittleEndian.PutUint64(h[:], slot*0x9e3779b97f4a7c15) + binary.LittleEndian.PutUint64(h[8:], ^slot) + h[31] = 0xa5 + return h +} + +// benchSlotHashes builds a 512-entry SlotHashes sysvar (newest first) that +// covers every slot the benchmark tower refers to. +func benchSlotHashes() SysvarSlotHashes { + const n = 512 + newest := benchVoteLastSlot + 1 + sh := make(SysvarSlotHashes, 0, n) + for i := uint64(0); i < n; i++ { + s := newest - i + sh = append(sh, SlotHash{Slot: s, Hash: benchSlotHash(s)}) + } + return sh +} + +// benchInitialVoteState is a fully populated current-version vote state with a +// full 31-entry tower ending at benchVoteLastSlot. +func benchInitialVoteState(voter solana.PublicKey) *VoteState { + vs := &VoteState{ + NodePubkey: voter, + AuthorizedWithdrawer: voter, + Commission: 10, + PriorVoters: PriorVoters{Index: 31, IsEmpty: true}, + EpochCredits: []EpochCredits{{Epoch: 0, Credits: 1000, PrevCredits: 0}}, + LastTimestamp: BlockTimestamp{Slot: benchVoteLastSlot, Timestamp: 1_700_000_000}, + } + vs.AuthorizedVoters.AuthorizedVoters.Set(0, voter) + root := benchVoteRoot + vs.RootSlot = &root + for i := uint64(0); i < MaxLockoutHistory; i++ { + vs.Votes.PushBack(LandedVote{ + Latency: 1, + Lockout: VoteLockout{Slot: benchVoteRoot + 1 + i, ConfirmationCount: uint32(MaxLockoutHistory - i)}, + }) + } + return vs +} + +func benchSerializedVoteState(vs *VoteState) []byte { + versioned := &VoteStateVersions{Type: VoteStateVersionCurrent, Current: *vs} + data := make([]byte, VoteStateV3Size) + if err := WriteVersionedVoteStateInPlace(data, versioned); err != nil { + panic(err) + } + return data +} + +// encodeTowerSync encodes a TowerSync that advances the tower by one slot: +// root = old root + 1, lockouts = old lockouts shifted by one slot plus the +// new slot, i.e. exactly what a validator sends every slot. +func encodeTowerSync() []byte { + root := benchVoteRoot + 1 + out := binary.LittleEndian.AppendUint32(nil, uint32(VoteProgramInstrTypeTowerSync)) + out = binary.LittleEndian.AppendUint64(out, root) + out = append(out, byte(MaxLockoutHistory)) // compact-u16, < 0x80 + prev := root + for i := uint64(0); i < MaxLockoutHistory; i++ { + slot := root + 1 + i + out = binary.AppendUvarint(out, slot-prev) + out = append(out, byte(MaxLockoutHistory-i)) + prev = slot + } + last := root + MaxLockoutHistory // benchVoteLastSlot + 1 + h := benchSlotHash(last) + out = append(out, h[:]...) + out = append(out, 1) // Some(timestamp) + out = binary.LittleEndian.AppendUint64(out, uint64(1_700_000_001)) + var blockID [32]byte + out = append(out, blockID[:]...) + return out +} + +type voteBench struct { + execCtx *ExecutionCtx + txAccts *TransactionAccounts + instr []byte + instrAccts []InstructionAccount + initial []byte +} + +func newVoteBench(skipTiming bool) *voteBench { + voter := benchPubkey(0x33) + votePk := benchPubkey(0x44) + initial := benchSerializedVoteState(benchInitialVoteState(voter)) + + voteProgramAcct := accounts.Account{Key: a.VoteProgramAddr, Lamports: 1, Data: []byte{}, Owner: a.NativeLoaderAddr, Executable: true} + voteAcct := accounts.Account{Key: votePk, Lamports: 1_000_000_000, Data: append([]byte(nil), initial...), Owner: a.VoteProgramAddr} + voterAcct := accounts.Account{Key: voter, Lamports: 1_000_000_000, Data: []byte{}, Owner: a.SystemProgramAddr} + txAccts := NewTransactionAccounts([]accounts.Account{voteProgramAcct, voteAcct, voterAcct}) + metas := []AccountMeta{ + {Pubkey: votePk, IsSigner: false, IsWritable: true}, + {Pubkey: voter, IsSigner: true, IsWritable: false}, + } + instrAccts := InstructionAcctsFromAccountMetas(metas, *txAccts) + + execCtx := newBenchExecCtx(txAccts, benchClockSlot, + features.EnableTowerSyncIx, + features.VoteStateAddVoteLatency, + features.TimelyVoteCredits, + features.DeprecateUnusedLegacyVotePlumbing, + ) + execCtx.SkipTimingMetrics = skipTiming + shAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarSlotHashesAddr, &shAcct); err != nil { + panic(err) + } + WriteSlotHashesSysvar(&execCtx.Accounts, benchSlotHashes()) + + return &voteBench{execCtx: execCtx, txAccts: txAccts, instr: encodeTowerSync(), instrAccts: instrAccts, initial: initial} +} + +// step runs one TowerSync against the initial vote state (the account data is +// rewound to the initial state first so every iteration performs identical work). +func (vb *voteBench) step() error { + copy(vb.txAccts.Accounts[1].Data, vb.initial) + resetTxCtx(vb.execCtx, vb.txAccts) + return vb.execCtx.ProcessInstruction(vb.instr, vb.instrAccts, []uint64{0}) +} + +func TestNativeVoteTowerSyncBenchSetup(t *testing.T) { + vb := newVoteBench(false) + if err := vb.step(); err != nil { + t.Fatalf("TowerSync failed: %v", err) + } + versioned, err := UnmarshalVersionedVoteState(vb.txAccts.Accounts[1].Data) + if err != nil { + t.Fatal(err) + } + vs := versioned.ConvertToCurrent() + if vs.Votes.Len() != MaxLockoutHistory { + t.Fatalf("tower length %d, want %d", vs.Votes.Len(), MaxLockoutHistory) + } + if last := vs.Votes.Back().Lockout.Slot; last != benchVoteLastSlot+1 { + t.Fatalf("last voted slot %d, want %d", last, benchVoteLastSlot+1) + } + if vs.RootSlot == nil || *vs.RootSlot != benchVoteRoot+1 { + t.Fatalf("root %v, want %d", vs.RootSlot, benchVoteRoot+1) + } +} + +func benchmarkVoteTowerSync(b *testing.B, skipTiming bool) { + vb := newVoteBench(skipTiming) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if err := vb.step(); err != nil { + b.Fatal(err) + } + } +} + +func BenchmarkNativeVoteTowerSync(b *testing.B) { benchmarkVoteTowerSync(b, false) } +func BenchmarkNativeVoteTowerSyncNoTiming(b *testing.B) { benchmarkVoteTowerSync(b, true) } + +// BenchmarkVoteStateRoundTrip isolates vote-state (de)serialization: decode +// the account, convert to current, re-encode — the fixed cost of every vote. +func BenchmarkVoteStateRoundTrip(b *testing.B) { + data := benchSerializedVoteState(benchInitialVoteState(benchPubkey(0x33))) + out := make([]byte, VoteStateV3Size) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + versioned, err := UnmarshalVersionedVoteState(data) + if err != nil { + b.Fatal(err) + } + vs := versioned.ConvertToCurrent() + cur := &VoteStateVersions{Type: VoteStateVersionCurrent, Current: *vs} + if err := WriteVersionedVoteStateInPlace(out, cur); err != nil { + b.Fatal(err) + } + } +} diff --git a/pkg/sealevel/program_cache_version_test.go b/pkg/sealevel/program_cache_version_test.go new file mode 100644 index 000000000..83089af25 --- /dev/null +++ b/pkg/sealevel/program_cache_version_test.go @@ -0,0 +1,72 @@ +package sealevel + +import ( + "bytes" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/features" + bin "github.com/gagliardetto/binary" + "github.com/stretchr/testify/require" + "testing" +) + +func TestExecutionRejectsOtherBankProgramCache(t *testing.T) { + for _, kind := range []string{"unbound", "same-slot-fork", "different-features"} { + t.Run(kind, func(t *testing.T) { + w := expandedProgramWorkloads(t)[0] + run := workloadRunner(t, w, false) + ctx, err := run() + require.NoError(t, err) + poison := &accountsdb.ProgramCacheEntry{DeploymentSlot: 1336} // nil executable: using it would panic + f := features.NewFeaturesDefault() + switch kind { + case "same-slot-fork": + poison.BindSource([]byte("another fork's program"), f) + case "different-features": + f.EnableFeature(features.VirtualAddressSpaceAdjustments, 0) + poison.BindSource(w.elf, f) + } + ctx.SlotCtx.AccountsDb.AddProgramToCache(w.program, poison) + next, err := run() + require.NoError(t, err) + w.check(t, next) + require.Equal(t, ctx.ComputeMeter.Used(), next.ComputeMeter.Used()) + }) + } +} + +func TestWarmCacheCannotBypassDeploymentSlot(t *testing.T) { + key, dataKey := benchPubkey(80), benchPubkey(81) + var encoded bytes.Buffer + state := UpgradeableLoaderState{Type: UpgradeableLoaderStateTypeProgram, Program: UpgradeableLoaderStateProgram{ProgramDataAddress: dataKey}} + require.NoError(t, state.MarshalWithEncoder(bin.NewBinEncoder(&encoded))) + program := accounts.Account{Key: key, Owner: a.BpfLoaderUpgradeableAddr, Executable: true, Lamports: 10000000, Data: encoded.Bytes()} + tx := NewTransactionAccounts([]accounts.Account{program}) + ctx := newBenchExecCtx(tx, 100) + initializeLegacyBankFixture(t, ctx) + ctx.SlotCtx.Slot = 100 + var data bytes.Buffer + state = UpgradeableLoaderState{Type: UpgradeableLoaderStateTypeProgramData, ProgramData: UpgradeableLoaderStateProgramData{Slot: 100}} + require.NoError(t, state.MarshalWithEncoder(bin.NewBinEncoder(&data))) + require.NoError(t, ctx.Accounts.SetAccount((*[32]byte)(&dataKey), &accounts.Account{Key: dataKey, Owner: a.BpfLoaderUpgradeableAddr, Lamports: 10000000, Data: data.Bytes()})) + // The cache describes a previous bank. Its older deployment slot must not + // hide this bank's deployment, which cannot be invoked in the same slot. + ctx.SlotCtx.AccountsDb.AddProgramToCache(dataKey, &accountsdb.ProgramCacheEntry{DeploymentSlot: 99}) + err := ctx.ProcessInstruction(nil, nil, []uint64{0}) + require.ErrorIs(t, err, InstrErrInvalidAccountData) +} + +func TestWarmCacheCannotBypassLoaderV4Retraction(t *testing.T) { + key := benchPubkey(82) + state := LoaderV4State{Slot: 99, Status: LoaderV4StatusRetracted} + program := accounts.Account{Key: key, Owner: a.LoaderV4Addr, Executable: true, Lamports: 10000000, Data: state.Marshal()} + tx := NewTransactionAccounts([]accounts.Account{program}) + ctx := newBenchExecCtx(tx, 100) + initializeLegacyBankFixture(t, ctx) + ctx.SlotCtx.Slot = 100 + require.NoError(t, ctx.Accounts.SetAccount((*[32]byte)(&key), &program)) + ctx.SlotCtx.AccountsDb.AddProgramToCache(key, &accountsdb.ProgramCacheEntry{DeploymentSlot: 99}) + err := ctx.ProcessInstruction(nil, nil, []uint64{0}) + require.ErrorIs(t, err, InstrErrUnsupportedProgramId) +} diff --git a/pkg/sealevel/program_workloads_bench_test.go b/pkg/sealevel/program_workloads_bench_test.go new file mode 100644 index 000000000..fd58e60b3 --- /dev/null +++ b/pkg/sealevel/program_workloads_bench_test.go @@ -0,0 +1,197 @@ +package sealevel + +import ( + "encoding/binary" + "os" + "path/filepath" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/Overclock-Validator/mithril/pkg/sbpf/loader" + "github.com/gagliardetto/solana-go" + "github.com/maypok86/otter" + "github.com/stretchr/testify/require" +) + +// External ELF inputs are pinned by SHA-256 in docs/sbpf-interpreter-benchmarks.md. +// No live account writes or network calls occur in this harness. Each invocation +// gets fresh account data; the program cache is warm and shared between runs. +type programWorkload struct { + name string + elf []byte + program solana.PublicKey + accts []accounts.Account + metas []AccountMeta + instruction []byte + check func(testing.TB, *ExecutionCtx) +} + +func expandedProgramWorkloads(t testing.TB) []programWorkload { + t.Helper() + program := benchPubkey(0x91) + from := accounts.Account{Key: benchPubkey(0x31), Owner: program, Lamports: 10000} + to := accounts.Account{Key: benchPubkey(0x32), Owner: program, Lamports: 10000} + cases := []programWorkload{{name: "BPF_LamportTransfer", elf: fixtures.Load(t, "sbpf", "cpi_c_to_bpf.so"), program: program, accts: []accounts.Account{from, to}, metas: []AccountMeta{{Pubkey: program}, {Pubkey: from.Key, IsSigner: true, IsWritable: true}, {Pubkey: to.Key, IsSigner: true, IsWritable: true}}, instruction: []byte{0}, check: func(t testing.TB, ctx *ExecutionCtx) { + src, err := ctx.TransactionContext.Accounts.GetAccount(1) + require.NoError(t, err) + dst, err := ctx.TransactionContext.Accounts.GetAccount(2) + require.NoError(t, err) + require.Equal(t, uint64(9000), src.Lamports) + require.Equal(t, uint64(11000), dst.Lamports) + }}} + + pda, bump, err := solana.FindProgramAddress([][]byte{[]byte("You pass butter")}, program) + require.NoError(t, err) + cases = append(cases, programWorkload{name: "CPI_Rust_SystemAllocate", elf: fixtures.Load(t, "sbpf", "cpi_rust_to_system_program_allocate.so"), program: program, + accts: []accounts.Account{{Key: a.SystemProgramAddr, Owner: a.NativeLoaderAddr, Executable: true, Lamports: 10000}, {Key: pda, Owner: a.SystemProgramAddr, Lamports: 10000}}, + metas: []AccountMeta{{Pubkey: a.SystemProgramAddr}, {Pubkey: pda, IsSigner: true, IsWritable: true}}, instruction: []byte{bump}, check: func(t testing.TB, ctx *ExecutionCtx) { + acct, e := ctx.TransactionContext.Accounts.GetAccount(2) + require.NoError(t, e) + require.Len(t, acct.Data, 1337) + require.NotEmpty(t, ctx.InnerInstrs) + }}) + dir := os.Getenv("MITHRIL_PROGRAM_BENCH_DIR") + if dir == "" { + return cases + } + arithmetic, err := os.ReadFile(filepath.Join(dir, "rotation_compute.so")) + require.NoError(t, err) + for _, iterations := range []uint32{500, 5000} { + n := iterations + data := append([]byte("RC01"), 0, 0, 0, 0) + binary.LittleEndian.PutUint32(data[4:], n) + name := "Arithmetic_500" + if n == 5000 { + name = "Arithmetic_5000" + } + cases = append(cases, programWorkload{name: name, elf: arithmetic, program: program, instruction: data, check: func(t testing.TB, ctx *ExecutionCtx) { + x := uint64(0x9e3779b97f4a7c15) + for i := uint32(0); i < n; i++ { + x = ((x << 7) | (x >> 57)) ^ (uint64(i) + 0x517cc1b727220a95) + } + _, got := ctx.TransactionContext.ReturnData() + require.Len(t, got, 8) + require.Equal(t, x, binary.LittleEndian.Uint64(got)) + }}) + } + token, err := os.ReadFile(filepath.Join(dir, "token2022.so")) + require.NoError(t, err) + mint, auth := benchPubkey(0x51), benchPubkey(0x52) + tokenData := func(amount uint64) []byte { + d := make([]byte, 165) + copy(d, mint[:]) + copy(d[32:], auth[:]) + binary.LittleEndian.PutUint64(d[64:], amount) + d[108] = 1 + return d + } + mintData := make([]byte, 82) + mintData[44] = 6 + mintData[45] = 1 + src := accounts.Account{Key: benchPubkey(0x53), Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: tokenData(1000000)} + dst := accounts.Account{Key: benchPubkey(0x54), Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: tokenData(5)} + instr := append([]byte{12}, binary.LittleEndian.AppendUint64(nil, 1000)...) + instr = append(instr, 6) + cases = append(cases, programWorkload{name: "Token2022_TransferChecked", elf: token, program: solana.Token2022ProgramID, + accts: []accounts.Account{src, {Key: mint, Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: mintData}, dst, {Key: auth, Owner: a.SystemProgramAddr, Lamports: 10000000}}, + metas: []AccountMeta{{Pubkey: src.Key, IsWritable: true}, {Pubkey: mint}, {Pubkey: dst.Key, IsWritable: true}, {Pubkey: auth, IsSigner: true}}, instruction: instr, + check: func(t testing.TB, ctx *ExecutionCtx) { + s, e := ctx.TransactionContext.Accounts.GetAccount(1) + require.NoError(t, e) + d, e := ctx.TransactionContext.Accounts.GetAccount(3) + require.NoError(t, e) + require.Equal(t, uint64(999000), binary.LittleEndian.Uint64(s.Data[64:])) + require.Equal(t, uint64(1005), binary.LittleEndian.Uint64(d.Data[64:])) + }}) + return cases +} +func workloadRunner(t testing.TB, w programWorkload, vasa bool) func() (*ExecutionCtx, error) { + t.Helper() + f := features.NewFeaturesDefault() + if vasa { + f.EnableFeature(features.VirtualAddressSpaceAdjustments, 0) + } + l, err := loader.NewLoaderWithSyscalls(w.elf, func(h uint32) (sbpf.Syscall, bool) { return Syscalls(f, false, h) }, false, f) + require.NoError(t, err) + prog, err := l.Load() + require.NoError(t, err) + require.NoError(t, prog.Verify()) + cache, err := otter.MustBuilder[solana.PublicKey, *accountsdb.ProgramCacheEntry](1024).Cost(func(solana.PublicKey, *accountsdb.ProgramCacheEntry) uint32 { return 1 }).Build() + require.NoError(t, err) + t.Cleanup(cache.Close) + db := &accountsdb.AccountsDb{ProgramCache: cache} + entry := &accountsdb.ProgramCacheEntry{Program: prog} + entry.BindSource(w.elf, f) + db.AddProgramToCache(w.program, entry) + _, cached := db.MaybeGetProgramFromCache(w.program) + require.True(t, cached, "warm program must be cached") + return func() (*ExecutionCtx, error) { + list := make([]accounts.Account, len(w.accts)+1) + list[0] = accounts.Account{Key: w.program, Owner: a.BpfLoader2Addr, Lamports: 10000000, Executable: true, Data: w.elf} + for i, acct := range w.accts { + list[i+1] = acct + list[i+1].Data = append([]byte(nil), acct.Data...) + } + tx := NewTransactionAccounts(list) + ctx := newBenchExecCtx(tx, 1337) + ctx.TransactionContext.ComputeBudgetLimits = &ComputeBudgetLimits{UpdatedHeapBytes: 32768} + ctx.Features = *f + ctx.ComputeMeter = cu.NewComputeMeter(1400000) + ctx.SlotCtx = &SlotCtx{Slot: 1337, AccountsDb: db} + ctx.Log = &LogRecorder{} + ctx.RecordInnerInstructions = true + err := ctx.ProcessInstruction(w.instruction, InstructionAcctsFromAccountMetas(w.metas, *tx), []uint64{0}) + return ctx, err + } +} +func TestProgramWorkloadResults(t *testing.T) { + for _, w := range expandedProgramWorkloads(t) { + for _, vasa := range []bool{false, true} { + name := w.name + if vasa { + name += "_VASA" + } + t.Run(name, func(t *testing.T) { + run := workloadRunner(t, w, vasa) + ctx, err := run() + require.NoError(t, err) + w.check(t, ctx) + t.Logf("cu=%d inner=%d", ctx.ComputeMeter.Used(), len(ctx.InnerInstrs)) + }) + } + } +} +func BenchmarkProgramWorkloads(b *testing.B) { + for _, w := range expandedProgramWorkloads(b) { + for _, vasa := range []bool{false, true} { + name := w.name + if vasa { + name += "_VASA" + } + b.Run(name, func(b *testing.B) { + run := workloadRunner(b, w, vasa) + ctx, err := run() + require.NoError(b, err) + w.check(b, ctx) + used := ctx.ComputeMeter.Used() + b.ReportAllocs() + b.ResetTimer() + for b.Loop() { + ctx, err = run() + if err != nil { + b.Fatal(err) + } + } + b.StopTimer() + w.check(b, ctx) + b.ReportMetric(float64(used), "cu/op") + }) + } + } +} diff --git a/pkg/sealevel/sealevel_bpf_loader_test.go b/pkg/sealevel/sealevel_bpf_loader_test.go index f20f13890..ab1b89d47 100644 --- a/pkg/sealevel/sealevel_bpf_loader_test.go +++ b/pkg/sealevel/sealevel_bpf_loader_test.go @@ -48,6 +48,7 @@ func TestExecute_Tx_BpfLoader_InitializeBuffer_Success(t *testing.T) { instrData := make([]byte, 4) binary.LittleEndian.AppendUint32(instrData, UpgradeableLoaderInstrTypeInitializeBuffer) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -91,6 +92,7 @@ func TestExecute_Tx_BpfLoader_InitializeBuffer_Buffer_Acct_Already_Initialize_Fa instrData := make([]byte, 4) binary.LittleEndian.AppendUint32(instrData, UpgradeableLoaderInstrTypeInitializeBuffer) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrAccountAlreadyInitialized, err) @@ -143,6 +145,7 @@ func TestExecute_Tx_BpfLoader_Write_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -203,6 +206,7 @@ func TestExecute_Tx_BpfLoader_Write_Offset_Too_Large_Failure(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrAccountDataTooSmall, err) } @@ -254,6 +258,7 @@ func TestExecute_Tx_BpfLoader_Write_Buffer_Authority_Didnt_Sign_Failure(t *testi txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -310,6 +315,7 @@ func TestExecute_Tx_BpfLoader_Write_Incorrect_Authority_Failure(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -346,8 +352,9 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Not_Enough_Instr_Accts_Failure(t *tes txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Success(t *testing.T) { @@ -391,6 +398,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -445,6 +453,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_ProgramData_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -499,6 +508,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Immutable_Failure(t *testing.T txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -549,6 +559,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Wrong_Upgrade_Authority_Failur txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -594,6 +605,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Authority_Didnt_Sign_Failure(t txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -633,6 +645,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_No_New_Authority_Failure(t *te txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -678,6 +691,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Uninitialized_Account_Failure( txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -723,6 +737,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_ProgramData_Immutable_Failure(t *test txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -768,6 +783,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_ProgramData_Authority_Didnt_Sign_Fail txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -818,6 +834,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_ProgramData_Wrong_Authority_Failure(t txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -855,9 +872,10 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Not_Enough_Instr_Accts_Failure txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Success(t *testing.T) { @@ -902,7 +920,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -958,7 +977,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Success(t *testing txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -1014,7 +1034,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Immutable_Failure(t *te txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -1066,7 +1087,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Wrong_Upgrade_Authority txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -1113,7 +1135,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Authority_Didnt_Sign_Fa txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1160,7 +1183,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_New_Authority_Didnt_Sig txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1207,7 +1231,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Uninitialized_Account_F txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -1254,7 +1279,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Immutable_Failure( txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -1301,7 +1327,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Authority_Didnt_Si txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1348,7 +1375,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_New_Authority_Didn txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1400,7 +1428,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Wrong_Authority_Fa txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -1446,6 +1475,7 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -1502,6 +1532,7 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Immutable_Failure(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -1547,6 +1578,7 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Authority_Didnt_Sign_Failure(t *testi txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1598,6 +1630,7 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Wrong_Authority_Failure(t *testing.T) txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -1636,6 +1669,7 @@ func TestExecute_Tx_BpfLoader_Close_Uninitialized_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -1680,6 +1714,7 @@ func TestExecute_Tx_BpfLoader_Close_Recipient_Same_As_Account_Being_Closed_Failu txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -1723,8 +1758,9 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Not_Enough_Accounts(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_BpfLoader_Close_ProgramData_Success(t *testing.T) { @@ -1787,6 +1823,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Success(t *testing.T) { clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -1861,8 +1898,9 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Not_Enough_Accounts_Failure(t *t clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_BpfLoader_Close_ProgramData_Program_Acct_Not_Writable_Failure(t *testing.T) { @@ -1925,6 +1963,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Program_Acct_Not_Writable_Failur clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -1990,6 +2029,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Program_Acct_Wrong_Owner_Failure clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectProgramId, err) } @@ -2055,6 +2095,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Already_Deployed_In_This_Block_F clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -2120,6 +2161,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_ProgramData_Not_A_Program_Acct_F clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -2185,6 +2227,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Nonclosable_Account_Failure(t *t clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -2265,6 +2308,7 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_Success(t *testing.T) { execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -2367,6 +2411,7 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_Extend_By_Zero_Bytes_Failure(t *test execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidInstructionData, err) } @@ -2447,6 +2492,7 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_With_Rent_Exemption_Payment_Not_Enou execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrNotEnoughAccountKeys, err) } @@ -2487,8 +2533,8 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_With_Rent_Exemption_Payment_Success( payerPrivateKey, err := solana.NewRandomPrivateKey() assert.NoError(t, err) payerPubkey := payerPrivateKey.PublicKey() - payerAcct := accounts.Account{Key: payerPubkey, Lamports: 10, Data: make([]byte, 0), Owner: a.SystemProgramAddr, Executable: false, RentEpoch: 100} - origPayerBalance := uint64(10) + payerAcct := accounts.Account{Key: payerPubkey, Lamports: 1_000_000, Data: make([]byte, 0), Owner: a.SystemProgramAddr, Executable: false, RentEpoch: 100} + origPayerBalance := uint64(1_000_000) // program account programPrivKey, err := solana.NewRandomPrivateKey() @@ -2540,6 +2586,7 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_With_Rent_Exemption_Payment_Success( execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -2666,10 +2713,11 @@ func TestExecute_Tx_BpfLoader_Upgrade_Success(t *testing.T) { rent.ExemptionThreshold = 1 rent.BurnPercent = 0 - rentAcct := accounts.Account{} + rentAcct := accounts.Account{Lamports: 1} execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -2777,10 +2825,11 @@ func TestExecute_Tx_BpfLoader_Upgrade_Buffer_Wrong_Authority_Failure(t *testing. rent.LamportsPerUint8Year = 1 rent.ExemptionThreshold = 1 rent.BurnPercent = 0 - rentAcct := accounts.Account{} + rentAcct := accounts.Account{Lamports: 1} execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -2885,6 +2934,7 @@ func TestExecute_Tx_BpfLoader_DeployWithMaxDataLen_Success(t *testing.T) { execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -2970,6 +3020,7 @@ func TestExecute_Tx_BpfLoader_Invoke_Bpf_Program_Success(t *testing.T) { execCtx.SlotCtx = new(SlotCtx) execCtx.SlotCtx.Slot = 1337 + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) } diff --git a/pkg/sealevel/sealevel_config_program_test.go b/pkg/sealevel/sealevel_config_program_test.go index 0941cd82c..3f33a5216 100644 --- a/pkg/sealevel/sealevel_config_program_test.go +++ b/pkg/sealevel/sealevel_config_program_test.go @@ -53,7 +53,7 @@ func TestExecute_Tx_Config_Program_Success(t *testing.T) { acct, err := txCtx.Accounts.GetAccount(1) require.NoError(t, err) - hasNewData := bytes.HasSuffix(acct.Data, []byte("bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb")) + hasNewData := bytes.Equal(acct.Data[:len(instrData)], instrData) assert.Equal(t, true, hasNewData) } diff --git a/pkg/sealevel/sealevel_system_program_test.go b/pkg/sealevel/sealevel_system_program_test.go index 952461fa7..73ce90e13 100644 --- a/pkg/sealevel/sealevel_system_program_test.go +++ b/pkg/sealevel/sealevel_system_program_test.go @@ -195,7 +195,7 @@ func TestExecute_Tx_System_Program_CreateAccount_Not_Enough_Accts_Failure(t *tes WriteRentSysvar(&execCtx.Accounts, rent) err = execCtx.ProcessInstruction(instrBytes, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_System_Program_CreateAccount_New_Acct_Has_Lamports_Failure(t *testing.T) { diff --git a/pkg/sealevel/sealevel_test.go b/pkg/sealevel/sealevel_test.go index 6ddfe230a..41ac410be 100644 --- a/pkg/sealevel/sealevel_test.go +++ b/pkg/sealevel/sealevel_test.go @@ -3,6 +3,7 @@ package sealevel import ( "bytes" _ "embed" + "encoding/binary" "encoding/json" "fmt" "io/fs" @@ -65,14 +66,17 @@ func TestInterpreter_Noop(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -105,13 +109,16 @@ func TestInterpreter_Memcpy_Strings_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.Equal(t, log.Logs, []string{ @@ -143,14 +150,17 @@ func TestInterpreter_Memcpy_Do_Not_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.Equal(t, log.Logs, []string{ @@ -181,14 +191,17 @@ func TestInterpreter_Memmove_Strings_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.Equal(t, log.Logs, []string{ @@ -220,14 +233,17 @@ func TestInterpreter_Memmove_Do_Not_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.Equal(t, log.Logs, []string{ @@ -257,14 +273,17 @@ func TestInterpreter_Memcpy_Overlapping(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() @@ -295,14 +314,17 @@ func TestInterpreter_Memcmp_Matches(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -336,14 +358,17 @@ func TestInterpreter_Memcmp_Does_Not_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -377,14 +402,17 @@ func TestInterpreter_Memset_Check_Correct(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -416,15 +444,18 @@ func TestInterpreter_Sha256(t *testing.T) { syscalls.Register("my_memcmp", SyscallMemcmp) var log LogRecorder + ctx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: ctx, + ComputeMeter: &ctx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -458,14 +489,17 @@ func TestInterpreter_Blake3(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -499,14 +533,17 @@ func TestInterpreter_Keccak256(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -541,14 +578,17 @@ func TestInterpreter_CreateProgramAddress(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -586,14 +626,17 @@ func TestInterpreter_TryFindProgramAddress(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -624,18 +667,24 @@ func TestInterpreter_TestPanic(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.Error(t, err) - assert.Equal(t, err.Error(), "exception at 16: SBF program Panicked in some_file_1234.c at 1337:10") + require.Contains(t, err.Error(), "SBF program Panicked in some_file_1234.c at 1337:10") + var exception *sbpf.Exception + require.ErrorAs(t, err, &exception) + require.Equal(t, int64(17), exception.PC) } func TestInterpreter_Secp256k1_Syscall(t *testing.T) { @@ -655,14 +704,17 @@ func TestInterpreter_Secp256k1_Syscall(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: syscalls, - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: syscalls, + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -733,7 +785,7 @@ func TestInterpreter_Get_Stack_Height_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -805,7 +857,7 @@ func TestInterpreter_ReturnData_Syscalls(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -890,132 +942,13 @@ func TestInterpreter_Poseidon_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) } -func TestInterpreter_Get_Sysvar_Syscalls(t *testing.T) { - // program data account - programDataPrivKey, err := solana.NewRandomPrivateKey() - assert.NoError(t, err) - programDataPubkey := programDataPrivKey.PublicKey() - programDataAcctState := UpgradeableLoaderState{Type: UpgradeableLoaderStateTypeProgramData, ProgramData: UpgradeableLoaderStateProgramData{Slot: 0, UpgradeAuthorityAddress: nil}} - validProgramBytes := fixtures.Load(t, "sbpf", "sysvars.so") - programDataStateWriter := new(bytes.Buffer) - programDataStateEncoder := bin.NewBinEncoder(programDataStateWriter) - err = programDataAcctState.MarshalWithEncoder(programDataStateEncoder) - assert.NoError(t, err) - programDataStateWriter.Write(validProgramBytes) - programDataStateBytes := make([]byte, len(validProgramBytes)+upgradeableLoaderSizeOfProgramDataMetaData) - copy(programDataStateBytes, programDataStateWriter.Bytes()) - copy(programDataStateBytes[upgradeableLoaderSizeOfProgramDataMetaData:], validProgramBytes) - - programDataAcct := accounts.Account{Key: programDataPubkey, Lamports: 0, Data: programDataStateBytes, Owner: a.BpfLoaderUpgradeableAddr, Executable: false, RentEpoch: 100} - - // program account - programAcctState := UpgradeableLoaderState{Type: UpgradeableLoaderStateTypeProgram, Program: UpgradeableLoaderStateProgram{ProgramDataAddress: programDataAcct.Key}} - programWriter := new(bytes.Buffer) - programEncoder := bin.NewBinEncoder(programWriter) - err = programAcctState.MarshalWithEncoder(programEncoder) - assert.NoError(t, err) - programBytes := programWriter.Bytes() - programPrivKey, err := solana.NewRandomPrivateKey() - assert.NoError(t, err) - programPubkey := programPrivKey.PublicKey() - programData := make([]byte, 5000) - copy(programData, programBytes) - programAcct := accounts.Account{Key: programPubkey, Lamports: 10000, Data: programData, Owner: a.BpfLoaderUpgradeableAddr, Executable: true, RentEpoch: 100} - - instrData := make([]byte, 0) - - transactionAccts := NewTransactionAccounts([]accounts.Account{programAcct}) - - acctMetas := []AccountMeta{{Pubkey: programAcct.Key, IsSigner: false, IsWritable: false}} - - instructionAccts := InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) - - txCtx := NewTransactionCtx(*transactionAccts, 5, 64) - var log LogRecorder - execCtx := ExecutionCtx{Log: &log, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeter(10000000000)} - - execCtx.Accounts = accounts.NewMemAccounts() - var clock SysvarClock - clock.Slot = 1234 - clock.Epoch = 1111 - clock.EpochStartTimestamp = 2222 - clock.UnixTimestamp = 3 - clock.LeaderScheduleEpoch = 100000 - clockAcct := accounts.Account{} - clockAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) - WriteClockSysvar(&execCtx.Accounts, clock) - - var rent SysvarRent - rent.LamportsPerUint8Year = 12 - rent.ExemptionThreshold = 34 - rent.BurnPercent = 56 - - rentAcct := accounts.Account{} - rentAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) - WriteRentSysvar(&execCtx.Accounts, rent) - - var epochSchedule SysvarEpochSchedule - epochSchedule.SlotsPerEpoch = 1111 - epochSchedule.LeaderScheduleSlotOffset = 2222 - epochSchedule.Warmup = true - epochSchedule.FirstNormalEpoch = 4444 - epochSchedule.FirstNormalSlot = 5555 - - epochScheduleAcct := accounts.Account{} - epochScheduleAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarEpochScheduleAddr, &epochScheduleAcct) - WriteEpochScheduleSysvar(&execCtx.Accounts, epochSchedule) - - var lastRestartSlot SysvarLastRestartSlot - lastRestartSlot.LastRestartSlot = 989898 - lastRestartSlotAcct := accounts.Account{} - lastRestartSlotAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarLastRestartSlotAddr, &lastRestartSlotAcct) - WriteLastRestartSlotSysvar(&execCtx.Accounts, lastRestartSlot) - - var epochRewards SysvarEpochRewards - epochRewards.DistributionStartingBlockHeight = 1234 - epochRewards.NumPartitions = 4321 - copy(epochRewards.ParentBlockhash[:], "abaaaaaaaaaaaaaaaaaaaaaaaaaaaada") - epochRewards.TotalPoints.Lo = 0xffffffffffffffff - epochRewards.TotalPoints.Hi = 0xeeeeeeeeeeeeeeee - epochRewards.TotalRewards = 5656 - epochRewards.DistributedRewards = 6767 - epochRewards.Active = false - epochRewardsAcct := accounts.Account{} - epochRewardsAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarEpochRewardsAddr, &epochRewardsAcct) - WriteEpochRewardsSysvar(&execCtx.Accounts, epochRewards) - - f := features.NewFeaturesDefault() - f.EnableFeature(features.LastRestartSlotSysvar, 0) - f.EnableFeature(features.EnablePartitionedEpochReward, 0) - execCtx.Features = *f - - pk := [32]byte(programDataAcct.Key) - err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) - assert.NoError(t, err) - - execCtx.SlotCtx = new(SlotCtx) - execCtx.SlotCtx.Slot = 1337 - - err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, nil, err) - - for _, l := range log.Logs { - fmt.Printf("log: %s\n", l) - } -} - func TestInterpreter_AltBn128_Ops_Syscall(t *testing.T) { // program data account programDataPrivKey, err := solana.NewRandomPrivateKey() @@ -1060,7 +993,7 @@ func TestInterpreter_AltBn128_Ops_Syscall(t *testing.T) { var log LogRecorder execCtx := ExecutionCtx{Log: &log, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeter(10000000000)} - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 execCtx.SlotCtx.Accounts = accounts.NewMemAccounts() @@ -1186,7 +1119,7 @@ func TestInterpreter_Alloc_Free_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1250,7 +1183,7 @@ func TestInterpreter_Alt_Bn128_Compression_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1314,7 +1247,7 @@ func TestInterpreter_Validate_Point_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1378,7 +1311,7 @@ func TestInterpreter_Curve_Group_Ops_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1442,7 +1375,7 @@ func TestInterpreter_Curve_Multiscalar_Mul_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1506,7 +1439,7 @@ func TestInterpreter_Log_Data_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1581,7 +1514,7 @@ func TestInterpreter_Cpi_C_System_Program_Allocate(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1661,7 +1594,7 @@ func TestInterpreter_Cpi_Rust_System_Program_Allocate(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1743,7 +1676,7 @@ func TestInterpreter_Cpi_C_Bpf_Program_Call(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1835,7 +1768,7 @@ func executeFirstBpfProgramAndReturnExecCtx(t *testing.T, log *LogRecorder, acct err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1911,14 +1844,21 @@ func TestInterpreter_Get_Processed_Sibling_Instruction_Test(t *testing.T) { fmt.Printf("******** second program call is %s\n", programAcct.Key) + // The introspection ELF only reads siblings; it does not invoke Allocate. + // Execute that preceding sibling explicitly before asking for indices 0/1. + allocateData := make([]byte, 12) + binary.LittleEndian.PutUint32(allocateData, SystemProgramInstrTypeAllocate) + binary.LittleEndian.PutUint64(allocateData[4:], 16) + allocateAccounts := InstructionAcctsFromAccountMetas([]AccountMeta{{Pubkey: acctToAlloc.Key, IsSigner: true, IsWritable: true}}, execCtx.TransactionContext.Accounts) + require.NoError(t, execCtx.ProcessInstruction(allocateData, allocateAccounts, []uint64{2})) + err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{1}) - assert.NoError(t, err) + require.NoError(t, err) - // test that the program logs from the CPI'd program (which calls get_processed_sibling_instruction) - // are as expected - expected := fmt.Sprintf("Program log: ******** sibling instruction 0 program id: %s", programAcct.Key) + // Check the introspection program reports both completed top-level siblings. + expected := fmt.Sprintf("Program log: ******** sibling instruction 0 program id: %s", solana.PublicKey(a.SystemProgramAddr)) assert.Equal(t, expected, log.Logs[1]) - expected = fmt.Sprintf("Program log: ******** sibling instruction 0 instruction data: %s", reformatHexBytes(instrData)) + expected = fmt.Sprintf("Program log: ******** sibling instruction 0 instruction data: %s", reformatHexBytes(allocateData)) assert.Equal(t, expected, log.Logs[2]) expected = fmt.Sprintf("Program log: ******** sibling instruction 1 program id: %s", firstProgramAcct.Key) @@ -1964,7 +1904,7 @@ func TestInterpreter_Test_Memo_Program_With_LoaderV2(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1981,7 +1921,7 @@ func TestInterpreter_Test_Memo_Program_With_LoaderV2(t *testing.T) { instrData[1] = 0xff err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, nil, err) + assert.Equal(t, InstrErrInvalidInstructionData, err) expected = fmt.Sprintf("Program log: Signed by %s", signerPubkey) containsExpected = strings.HasPrefix(log.Logs[2], expected) @@ -2039,7 +1979,7 @@ func TestInterpreter_Test_Deprecated_Loader(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -2084,9 +2024,13 @@ func (e *executeCase) run(t *testing.T) { tx.PushInstructionCtx(InstructionCtx{}) opts := tx.newVMOpts(&e.Params) opts.Tracer = testLogger{t} + ctx := opts.Context.(*ExecutionCtx) + ctx.ComputeMeter = cu.NewComputeMeter(uint64(opts.MaxCU)) + opts.ComputeMeter = &ctx.ComputeMeter interpreter := sbpf.NewInterpreter(program, opts) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.NoError(t, err) diff --git a/pkg/sealevel/spl_token_demo_test.go b/pkg/sealevel/spl_token_demo_test.go index c0ede8c86..0a6cc6441 100644 --- a/pkg/sealevel/spl_token_demo_test.go +++ b/pkg/sealevel/spl_token_demo_test.go @@ -24,7 +24,7 @@ var splTokenProgramAddr = base58.MustDecodeFromString("TokenkegQfeZyiNwAJbNbGKPF // spl token program later. func setupSplTokenProgramAccount(t *testing.T, accts *accounts.Accounts) accounts.Account { programBytes := fixtures.Load(t, "sbpf", "spl-token.so") - splTokenAcct := accounts.Account{Key: splTokenProgramAddr, Lamports: 0, Data: programBytes, Owner: a.BpfLoader2Addr, Executable: true, RentEpoch: 100} + splTokenAcct := accounts.Account{Key: splTokenProgramAddr, Lamports: 1, Data: programBytes, Owner: a.BpfLoader2Addr, Executable: true, RentEpoch: 100} pk := [32]byte(splTokenProgramAddr) err := (*accts).SetAccount(&pk, &splTokenAcct) @@ -201,6 +201,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { {Pubkey: SysvarRentAddr, IsSigner: false, IsWritable: false}} instructionAccts := InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // InitializeMint: execute SPL token InitializeMint instruction err := execCtx.ProcessInstruction(initMintInstrData, instructionAccts, []uint64{0}) @@ -231,6 +232,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { instructionAccts = InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // InitializeAccount: execute SPL token InitializeMint instruction err = execCtx.ProcessInstruction(initAccountInstrData, instructionAccts, []uint64{0}) @@ -260,6 +262,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { instructionAccts = InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // InitializeAccount: execute SPL token InitializeMint instruction err = execCtx.ProcessInstruction(initAccountInstrData, instructionAccts, []uint64{0}) @@ -281,6 +284,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { instructionAccts = InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // MintTo: serialize up a MintTo instruction numTokensToMint := uint64(61616161) @@ -306,6 +310,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { instructionAccts = InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // Transfer: serialize up a Transfer instruction numTokensToTransfer := uint64(1337) diff --git a/pkg/sealevel/syscalls_call.go b/pkg/sealevel/syscalls_call.go index 0ffe7b0c3..193f44baa 100644 --- a/pkg/sealevel/syscalls_call.go +++ b/pkg/sealevel/syscalls_call.go @@ -53,11 +53,11 @@ func SyscallGetReturnDataImpl(vm sbpf.VM, returnDataAddr, length, programIdAddr return syscallErr(err) } - if len(returnData) != len(returnDataResult) { + if int(length) != len(returnDataResult) { return syscallErr(SyscallErrInvalidLength) } - copy(returnDataResult, returnData) + copy(returnDataResult, returnData[:length]) var programIdResult []byte programIdResult, err = vm.Translate(programIdAddr, solana.PublicKeyLength, true) @@ -170,7 +170,7 @@ func SyscallGetProcessedSiblingInstructionImpl(vm sbpf.VM, index, metaAddr, prog } if instrCtxFound != nil { - resultsHeaderBytes, err := vm.Translate(metaAddr, ProcessedSiblingInstructionSize, false) + resultsHeaderBytes, err := vm.Translate(metaAddr, ProcessedSiblingInstructionSize, true) if err != nil { return syscallErr(err) } diff --git a/pkg/sealevel/syscalls_hash.go b/pkg/sealevel/syscalls_hash.go index b73a1cd0c..33edd31d0 100644 --- a/pkg/sealevel/syscalls_hash.go +++ b/pkg/sealevel/syscalls_hash.go @@ -3,6 +3,7 @@ package sealevel import ( "bytes" "crypto/sha256" + "encoding/binary" "fmt" "math/big" @@ -49,15 +50,13 @@ func SyscallSha256Impl(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint6 } var data []byte - reader := bytes.NewReader(vals) + // Translate validated the complete descriptor array above. Decode directly + // to avoid allocating a reader and a temporary buffer for each slice. for count := uint64(0); count < valsLen; count++ { - var vec VectorDescrC - err = vec.Unmarshal(reader) - if err != nil { - return syscallErr(err) - } + offset := count * 16 + vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} data, err = vm.Translate(vec.Addr, vec.Len, false) if err != nil { @@ -73,7 +72,9 @@ func SyscallSha256Impl(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint6 hasher.Write(data) } } - copy(hashResult[:], hasher.Sum(nil)) + // All inputs have been read before writing, including when output aliases + // input memory. Append into the translated destination without allocating. + hasher.Sum(hashResult[:0]) return syscallSuccess(0) } diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 7bea5331e..14a1e39b0 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -1,6 +1,7 @@ package sealevel import ( + "bytes" "encoding/binary" //"github.com/Overclock-Validator/mithril/pkg/mlog" @@ -15,14 +16,29 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { return execCtx.ComputeMeter.Consume(cost) } -func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) (err error) { - srcBuf := make([]byte, n) - err = vm.Read(src, srcBuf) +// memmoveImplInternal copies n bytes from src to dst inside the VM without a +// temporary buffer. The source is translated first so a bad source address is +// reported before a bad destination, as before. Go's copy has memmove +// semantics, so overlapping ranges within one region are handled; and when +// the destination translation grows or copy-on-writes an account region, the +// source slice still refers to the previous backing buffer, whose bytes are +// exactly what the old read-then-write sequence would have copied. +func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { + // Agave's touch_slice_mut / translate_slice return empty slices before + // address lookup for zero length. CU was already charged by the syscall. + if n == 0 { + return nil + } + srcMem, err := vm.Translate(src, n, false) if err != nil { - return + return err + } + dstMem, err := vm.Translate(dst, n, true) + if err != nil { + return err } - err = vm.Write(dst, srcBuf) - return + copy(dstMem, srcMem) + return nil } // SyscallMemcpyImpl is the implementation of the memcpy (sol_memcpy_) syscall. @@ -76,6 +92,26 @@ func SyscallMemmoveImpl(vm sbpf.VM, dst, src, n uint64) (uint64, error) { var SyscallMemmove = sbpf.SyscallFunc3(SyscallMemmoveImpl) +// memcmpResult returns the C memcmp result of two equal-length slices: zero +// when they are equal, otherwise the difference of the first differing bytes +// as unsigned values, matching Agave's `(b1 as i32) - (b2 as i32)`. +func memcmpResult(a, b []byte) int32 { + if bytes.Equal(a, b) { + return 0 + } + // The slices differ: skip equal 8-byte words, then locate the byte. + i := 0 + for i+8 <= len(a) && binary.LittleEndian.Uint64(a[i:]) == binary.LittleEndian.Uint64(b[i:]) { + i += 8 + } + for ; i < len(a); i++ { + if a[i] != b[i] { + return int32(a[i]) - int32(b[i]) + } + } + return 0 +} + // SyscallMemcmpImpl is the implementation for the memcmp (sol_memcmp_) syscall. func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, error) { //mlog.Log.Debugf("SyscallMemcmp") @@ -96,15 +132,7 @@ func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, return syscallErr(err) } - cmpResult := int32(0) - for count := uint64(0); count < n; count++ { - b1 := slice1[count] - b2 := slice2[count] - if b1 != b2 { - cmpResult = int32(b1) - int32(b2) - break - } - } + cmpResult := memcmpResult(slice1, slice2) resultSlice, err := vm.Translate(resultAddr, 4, true) if err != nil { @@ -118,7 +146,23 @@ func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, var SyscallMemcmp = sbpf.SyscallFunc4(SyscallMemcmpImpl) -// SyscallMemcmpImpl is the implementation for the memset (sol_memset_) syscall. +// memsetBytes fills mem with c using the runtime's block clear for zero and a +// doubling copy otherwise, instead of a byte-at-a-time loop. +func memsetBytes(mem []byte, c byte) { + if len(mem) == 0 { + return + } + if c == 0 { + clear(mem) + return + } + mem[0] = c + for filled := 1; filled < len(mem); filled *= 2 { + copy(mem[filled:], mem[:filled]) + } +} + +// SyscallMemsetImpl is the implementation for the memset (sol_memset_) syscall. func SyscallMemsetImpl(vm sbpf.VM, dst, c, n uint64) (uint64, error) { //mlog.Log.Debugf("SyscallMemset") @@ -133,16 +177,14 @@ func SyscallMemsetImpl(vm sbpf.VM, dst, c, n uint64) (uint64, error) { return syscallErr(err) } - for i := uint64(0); i < n; i++ { - mem[i] = byte(c) - } + memsetBytes(mem, byte(c)) return syscallSuccess(0) } var SyscallMemset = sbpf.SyscallFunc3(SyscallMemsetImpl) -// SyscallMemcmpImpl is the implementation for the memset (sol_memset_) syscall. +// SyscallAllocFreeImpl is the implementation for the alloc/free (sol_alloc_free_) syscall. func SyscallAllocFreeImpl(vm sbpf.VM, size, freeAddr uint64) (uint64, error) { //mlog.Log.Debugf("SyscallAllocFreeImpl") diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go new file mode 100644 index 000000000..d375a42b7 --- /dev/null +++ b/pkg/sealevel/syscalls_mem_test.go @@ -0,0 +1,248 @@ +package sealevel + +import ( + "bytes" + "encoding/binary" + "math/rand" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + feat "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +func newMemSyscallVM(t *testing.T, input []byte, regions []sbpf.InputRegion) (*sbpf.Interpreter, *ExecutionCtx) { + t.Helper() + features := feat.NewFeaturesDefault() + execCtx := &ExecutionCtx{Features: *features, ComputeMeter: cu.NewComputeMeter(1_000_000)} + vm := sbpf.NewInterpreter(&sbpf.Program{TextVA: sbpf.VaddrProgram, Funcs: map[uint32]int64{}}, &sbpf.VMOpts{ + Input: input, + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, + InputRegions: regions, + }) + t.Cleanup(vm.Finish) + return vm, execCtx +} + +func TestSyscallMemmoveOverlapping(t *testing.T) { + for _, test := range []struct { + name string + dst, src, n uint64 + expectMemcpyE bool + }{ + {name: "forward-overlap", dst: 4, src: 0, n: 16, expectMemcpyE: true}, + {name: "backward-overlap", dst: 0, src: 4, n: 16, expectMemcpyE: true}, + {name: "disjoint", dst: 40, src: 0, n: 16}, + {name: "adjacent", dst: 16, src: 0, n: 16}, + {name: "empty", dst: 0, src: 0, n: 0}, + } { + t.Run(test.name, func(t *testing.T) { + input := make([]byte, 64) + for i := range input { + input[i] = byte(i + 1) + } + want := append([]byte(nil), input...) + copy(want[test.dst:test.dst+test.n], want[test.src:test.src+test.n]) + + vm, _ := newMemSyscallVM(t, input, nil) + ret, err := SyscallMemmoveImpl(vm, sbpf.VaddrInput+test.dst, sbpf.VaddrInput+test.src, test.n) + require.NoError(t, err) + require.Zero(t, ret) + require.Equal(t, want, input, "memmove must have Go copy (memmove) semantics") + + // memcpy: the same result for disjoint ranges, an error for overlap. + input2 := make([]byte, 64) + for i := range input2 { + input2[i] = byte(i + 1) + } + vm2, _ := newMemSyscallVM(t, input2, nil) + _, err = SyscallMemcpyImpl(vm2, sbpf.VaddrInput+test.dst, sbpf.VaddrInput+test.src, test.n) + if test.expectMemcpyE { + require.ErrorIs(t, err, SyscallErrCopyOverlapping) + } else { + require.NoError(t, err) + require.Equal(t, want, input2) + } + }) + } +} + +func TestSyscallMemmoveBadAddressOrder(t *testing.T) { + input := make([]byte, 32) + vm, _ := newMemSyscallVM(t, input, nil) + // Unreadable source is reported before an unwritable destination. + _, err := SyscallMemmoveImpl(vm, sbpf.VaddrProgram, sbpf.VaddrInput+100, 8) + require.Error(t, err) + var badAccess sbpf.ExcBadAccess + require.ErrorAs(t, err, &badAccess) + require.False(t, badAccess.Write, "the source translation must fail first") + // Readable source, write to a read-only region. + _, err = SyscallMemmoveImpl(vm, sbpf.VaddrProgram, sbpf.VaddrInput, 8) + require.Error(t, err) + require.ErrorAs(t, err, &badAccess) + require.True(t, badAccess.Write) +} + +func TestSyscallMemmoveIntoGrowingInputRegion(t *testing.T) { + // The destination region copy-on-writes and grows on first write; the + // source slice taken before that must still yield the original bytes. + original := []byte{1, 2, 3, 4, 5, 6, 7, 8} + var replaced []byte + region := sbpf.InputRegion{ + Offset: 0, + RegionSize: uint64(len(original)), + AddressSpaceReserved: 32, + Writable: false, + AccountIndex: 0, + Data: original, + OnWrite: func(region *sbpf.InputRegion, requestedLen uint64) error { + replaced = make([]byte, 32) + copy(replaced, region.Data) + region.Data = replaced + region.RegionSize = 32 + region.Writable = true + return nil + }, + } + vm, _ := newMemSyscallVM(t, nil, []sbpf.InputRegion{region}) + // Copy the first 4 bytes over bytes 4..8 within the same region: the + // source translation sees the original buffer, the destination the clone. + _, err := SyscallMemmoveImpl(vm, sbpf.VaddrInput+4, sbpf.VaddrInput, 4) + require.NoError(t, err) + require.NotNil(t, replaced, "the first write must trigger the copy-on-write hook") + require.Equal(t, []byte{1, 2, 3, 4, 1, 2, 3, 4}, replaced[:8]) + require.Equal(t, []byte{1, 2, 3, 4, 5, 6, 7, 8}, original, "the shared buffer must stay untouched") +} + +func TestSyscallMemcmpAndMemset(t *testing.T) { + input := make([]byte, 128) + for i := range input { + input[i] = byte(i) + } + vm, _ := newMemSyscallVM(t, input, nil) + + // memcmp of equal and differing 32-byte keys, result written at 96. + copy(input[32:64], input[0:32]) + _, err := SyscallMemcmpImpl(vm, sbpf.VaddrInput, sbpf.VaddrInput+32, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(0), int32(binary.LittleEndian.Uint32(input[96:]))) + input[63] = 0xff + _, err = SyscallMemcmpImpl(vm, sbpf.VaddrInput, sbpf.VaddrInput+32, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(31)-int32(0xff), int32(binary.LittleEndian.Uint32(input[96:]))) + _, err = SyscallMemcmpImpl(vm, sbpf.VaddrInput+32, sbpf.VaddrInput, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(0xff)-int32(31), int32(binary.LittleEndian.Uint32(input[96:]))) + + // memset 0xab over 33 bytes, then zero over 9 bytes. + sentinel := input[97] // The preceding memcmp result overwrote bytes 96..99. + _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+64, 0x1ab, 33) + require.NoError(t, err) + require.Equal(t, bytes.Repeat([]byte{0xab}, 33), input[64:97]) + require.Equal(t, sentinel, input[97]) + _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+70, 0, 9) + require.NoError(t, err) + require.Equal(t, bytes.Repeat([]byte{0xab}, 6), input[64:70]) + require.Equal(t, make([]byte, 9), input[70:79]) + require.Equal(t, bytes.Repeat([]byte{0xab}, 18), input[79:97]) +} + +// The byte loop the syscalls used before memcmpResult; kept as the reference. +func referenceMemcmp(a, b []byte) int32 { + for i := range a { + if a[i] != b[i] { + return int32(a[i]) - int32(b[i]) + } + } + return 0 +} + +func TestMemcmpResultMatchesByteLoop(t *testing.T) { + rng := rand.New(rand.NewSource(7)) + for iter := 0; iter < 100000; iter++ { + n := rng.Intn(70) + a := make([]byte, n) + rng.Read(a) + b := append([]byte(nil), a...) + if n > 0 && rng.Intn(4) != 0 { + b[rng.Intn(n)] = byte(rng.Intn(256)) + if rng.Intn(2) == 0 { + b[rng.Intn(n)] ^= byte(1 + rng.Intn(255)) + } + } + if got, want := memcmpResult(a, b), referenceMemcmp(a, b); got != want { + t.Fatalf("n=%d a=%x b=%x: got %d want %d", n, a, b, got, want) + } + } + if memcmpResult([]byte{0xff}, []byte{0x00}) != 255 || memcmpResult([]byte{0x00}, []byte{0xff}) != -255 { + t.Fatal("memcmp must return the unsigned byte difference") + } + if memcmpResult(nil, nil) != 0 { + t.Fatal("empty compare must be 0") + } +} + +func TestMemsetBytes(t *testing.T) { + for _, n := range []int{0, 1, 2, 3, 7, 8, 9, 31, 32, 33, 100, 1023, 4096, 10001} { + for _, c := range []byte{0, 1, 0x7f, 0xff} { + mem := make([]byte, n) + for i := range mem { + mem[i] = byte(i) + } + memsetBytes(mem, c) + if !bytes.Equal(mem, bytes.Repeat([]byte{c}, n)) { + t.Fatalf("n=%d c=%d: %x", n, c, mem) + } + } + } +} + +func TestSyscallMemoryZeroLengthSkipsAddressValidation(t *testing.T) { + for _, src := range []uint64{sbpf.VaddrInput, 0, ^uint64(0)} { + for _, dst := range []uint64{sbpf.VaddrInput, sbpf.VaddrProgram, ^uint64(0)} { + for _, fn := range []func(sbpf.VM, uint64, uint64, uint64) (uint64, error){SyscallMemmoveImpl, SyscallMemcpyImpl} { + input := bytes.Repeat([]byte{0x42}, 32) + vm, ctx := newMemSyscallVM(t, input, nil) + before := ctx.ComputeMeter.Remaining() + ret, err := fn(vm, dst, src, 0) + require.NoError(t, err) + require.Zero(t, ret) + require.Equal(t, before-cu.CUMemOpBaseCost, ctx.ComputeMeter.Remaining()) + require.Equal(t, bytes.Repeat([]byte{0x42}, 32), input) + } + } + } +} + +// Compare the syscall with the old read-then-write path, including charged CU. +func TestMemoryCopyDifferential(t *testing.T) { + rng := rand.New(rand.NewSource(127)) + for i := 0; i < 300; i++ { + data := make([]byte, 128) + rng.Read(data) + actual, want := append([]byte(nil), data...), append([]byte(nil), data...) + vm, ctx := newMemSyscallVM(t, actual, nil) + ref, refCtx := newMemSyscallVM(t, want, nil) + src, dst, n := uint64(rng.Intn(145)), uint64(rng.Intn(145)), uint64(rng.Intn(80)) + src += sbpf.VaddrInput + dst += sbpf.VaddrInput + _, gotErr := SyscallMemmoveImpl(vm, dst, src, n) + wantErr := MemOpConsume(refCtx, n) + if wantErr == nil && n > 0 { + buf := make([]byte, n) + wantErr = ref.Read(src, buf) + if wantErr == nil { + wantErr = ref.Write(dst, buf) + } + } + if wantErr == nil { + require.NoError(t, gotErr) + } else { + require.EqualError(t, gotErr, wantErr.Error()) + } + require.Equal(t, want, actual) + require.Equal(t, refCtx.ComputeMeter.Remaining(), ctx.ComputeMeter.Remaining()) + } +} diff --git a/pkg/sealevel/syscalls_return_data_test.go b/pkg/sealevel/syscalls_return_data_test.go new file mode 100644 index 000000000..e3252a154 --- /dev/null +++ b/pkg/sealevel/syscalls_return_data_test.go @@ -0,0 +1,43 @@ +package sealevel + +import ( + "bytes" + "fmt" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" + "testing" +) + +func TestSyscallGetReturnDataPrefix(t *testing.T) { + for _, n := range []uint64{0, 1, 7, 32, 64} { + t.Run(fmt.Sprint(n), func(t *testing.T) { + input := bytes.Repeat([]byte{0xCC}, 128) + vm, ctx := newMemSyscallVM(t, input, nil) + ctx.TransactionContext = &TransactionCtx{} + data := bytes.Repeat([]byte{0x5A}, 32) + pk := solana.PublicKey{0x72} + ctx.TransactionContext.SetReturnData(pk, data) + before := ctx.ComputeMeter.Remaining() + got, err := SyscallGetReturnDataImpl(vm, sbpf.VaddrInput, n, sbpf.VaddrInput+64) + require.NoError(t, err) + require.Equal(t, uint64(len(data)), got) + copied := min(n, uint64(len(data))) + actual, err := vm.Translate(sbpf.VaddrInput, 128, false) + require.NoError(t, err) + require.Equal(t, data[:copied], actual[:copied]) + require.Equal(t, bytes.Repeat([]byte{0xCC}, int(64-copied)), actual[copied:64]) + charge := uint64(cu.CUSyscallBaseCost) + if copied != 0 { + require.Equal(t, pk[:], actual[64:96]) + charge += (copied + 32) / cu.CUCpiBytesPerUnit + } else { + require.Equal(t, bytes.Repeat([]byte{0xCC}, 32), actual[64:96]) + } + require.Equal(t, charge, before-ctx.ComputeMeter.Remaining()) + _, retained := ctx.TransactionContext.ReturnData() + require.Equal(t, data, retained) + }) + } +} diff --git a/pkg/sealevel/syscalls_sha256_bench_test.go b/pkg/sealevel/syscalls_sha256_bench_test.go new file mode 100644 index 000000000..6c972f37f --- /dev/null +++ b/pkg/sealevel/syscalls_sha256_bench_test.go @@ -0,0 +1,191 @@ +package sealevel + +import ( + "bytes" + "crypto/sha256" + "encoding/binary" + "fmt" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" + "math/rand" + "testing" +) + +// Frozen syscall implementation from bd17683a; keep independent for differential tests. +func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { + if valsLen > cu.CUSha256MaxSlices { + return syscallErr(SyscallErrTooManySlices) + } + + execCtx := executionCtx(vm) + err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) + if err != nil { + return syscallCuErr() + } + + hashResult, err := vm.Translate(resultsAddr, 32, true) + if err != nil { + return syscallErr(err) + } + + hasher := sha256.New() + if valsLen > 0 { + var vals []byte + + // The data at 'valsAddr' consists of an array of 'slice references', which consists + // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that + // refers to an input value to hash. + // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above + vals, err = vm.Translate(valsAddr, valsLen*16, false) + if err != nil { + return syscallErr(err) + } + + var data []byte + reader := bytes.NewReader(vals) + + for count := uint64(0); count < valsLen; count++ { + + var vec VectorDescrC + err = vec.Unmarshal(reader) + if err != nil { + return syscallErr(err) + } + + data, err = vm.Translate(vec.Addr, vec.Len, false) + if err != nil { + return syscallErr(err) + } + + cost := max(vec.Len/2, cu.CUMemOpBaseCost) + err = execCtx.ComputeMeter.Consume(cost) + if err != nil { + return syscallCuErr() + } + + hasher.Write(data) + } + } + copy(hashResult[:], hasher.Sum(nil)) + return syscallSuccess(0) +} + +type sha256Call func(sbpf.VM, uint64, uint64, uint64) (uint64, error) + +func sha256Fixture(sizes []int) ([]byte, uint64, uint64, uint64) { + mem := make([]byte, 32768) + pos := 8192 + for i, n := range sizes { + binary.LittleEndian.PutUint64(mem[i*16:], sbpf.VaddrInput+uint64(pos)) + binary.LittleEndian.PutUint64(mem[i*16+8:], uint64(n)) + for j := 0; j < n; j++ { + mem[pos+j] = byte(i + j) + } + pos += n + } + return mem, sbpf.VaddrInput, uint64(len(sizes)), sbpf.VaddrInput + 4096 +} + +func sha256VM(mem []byte, budget uint64) (*sbpf.Interpreter, *ExecutionCtx) { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(budget)} + vm := sbpf.NewInterpreter(&sbpf.Program{}, &sbpf.VMOpts{Input: mem, HeapMax: 32768, Context: ctx, ComputeMeter: &ctx.ComputeMeter}) + return vm, ctx +} + +func TestSha256SyscallDifferential(t *testing.T) { + rng := rand.New(rand.NewSource(1234)) + for i := 0; i < 700; i++ { + sizes := make([]int, rng.Intn(8)) + for j := range sizes { + sizes[j] = rng.Intn(80) + } + if i < 8 { + sizes = [][]int{nil, {0}, {32, 4}, {55}, {56}, {32, 24}, {4096}, {0, 32, 0, 4}}[i] + } + mem, a, n, out := sha256Fixture(sizes) + budget := uint64(100000) + switch i % 11 { + case 1: + budget = uint64(rng.Intn(250)) + case 2: + out = 1 + case 3: + a = 1 + case 4: + if n > 0 { + binary.LittleEndian.PutUint64(mem, 1) + } + case 5: + if n > 0 { + binary.LittleEndian.PutUint64(mem[8:], ^uint64(0)) + } + case 6: + out = sbpf.VaddrInput + 8192 // output overlaps the input + case 7: + out = a // output overlaps descriptors + case 8: + n = cu.CUSha256MaxSlices + 1 + case 9: + n = cu.CUSha256MaxSlices + case 10: + a = sbpf.VaddrInput + uint64(len(mem)-1) + } + var wantMem []byte + var wantRet, wantCU uint64 + var wantErr string + for k, fn := range []sha256Call{sha256BaselineReference, SyscallSha256Impl} { + buf := append([]byte(nil), mem...) + vm, ctx := sha256VM(buf, budget) + ret, err := fn(vm, a, n, out) + remaining := ctx.ComputeMeter.Remaining() + vm.Finish() + if k == 0 { + wantMem = buf + wantRet = ret + wantCU = remaining + wantErr = fmt.Sprint(err) + continue + } + require.Equal(t, wantRet, ret, "case %d variant %d", i, k) + require.Equal(t, wantErr, fmt.Sprint(err), "case %d variant %d", i, k) + require.Equal(t, wantCU, remaining, "case %d variant %d", i, k) + require.Equal(t, wantMem, buf, "case %d variant %d", i, k) + } + } +} + +var sha256BenchDigest [32]byte + +func BenchmarkSha256Syscall(b *testing.B) { + for _, tc := range []struct { + name string + sizes []int + }{{"empty", nil}, {"36_contiguous", []int{36}}, {"32_plus_4", []int{32, 4}}, {"55", []int{55}}, {"56", []int{56}}, {"1232", []int{1232}}, {"4096", []int{4096}}} { + for _, variant := range []struct { + name string + fn sha256Call + }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}} { + b.Run(tc.name+"/"+variant.name, func(b *testing.B) { + mem, a, n, out := sha256Fixture(tc.sizes) + vm, ctx := sha256VM(mem, ^uint64(0)) + defer vm.Finish() + b.ReportAllocs() + b.ResetTimer() + for j := 0; j < b.N; j++ { + ctx.ComputeMeter = cu.NewComputeMeter(100000) + if _, err := variant.fn(vm, a, n, out); err != nil { + b.Fatal(err) + } + } + }) + } + } + b.Run("raw36", func(b *testing.B) { + var data [36]byte + b.ReportAllocs() + for j := 0; j < b.N; j++ { + sha256BenchDigest = sha256.Sum256(data[:]) + } + }) +} diff --git a/pkg/sealevel/syscalls_sha256_loop_test.go b/pkg/sealevel/syscalls_sha256_loop_test.go new file mode 100644 index 000000000..32c22d9f0 --- /dev/null +++ b/pkg/sealevel/syscalls_sha256_loop_test.go @@ -0,0 +1,102 @@ +package sealevel + +import ( + "crypto/sha256" + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +// This is text slots 518..545 from the captured AogGeA81 program's hash loop. +// Captured ELF SHA-256: b3286f96f5611ee7db62dc11ea9aab1afe80908c08a0fbd3fdd783879d70ed88. +// The captured ELF uses SBF v0. Only the unresolved syscall relocation is replaced. The harness supplies a +// loop bound and zero initial state; transaction loading, CPI and the rest of +// the program are deliberately excluded. +func sha256LoopProgram(iterations uint32) *sbpf.Program { + text := []sbpf.Slot{ + sbpf.Slot(sbpf.OpMov64Imm) | 7<<8 | sbpf.Slot(iterations)<<32, + 0xa1bf, 0xffffffb000000107, 0xfe701a7b, 0xa1bf, 0xfffffe1000000107, + 0xfe601a7b, 0xffb08a63, 0x4fe780a7a, 0x20fe680a7a, 0xa1bf, + 0xfffffe6000000107, 0xa3bf, 0xffffffe000000307, 0x2000002b7, + sbpf.Slot(sbpf.OpCall) | sbpf.Slot(hash_sol_sha256)<<32, + 0xffe0a179, 0xfe101a7b, 0xffe8a179, 0xfe181a7b, 0xfff0a179, + 0xfe201a7b, 0xfff8a179, 0xfe281a7b, 0x100000807, 0x81bf, + 0x2000000167, 0x2000000177, 0xffe471ad, + // Return the first digest word so the harness can check the computation. + 0xfe10a079, sbpf.Slot(sbpf.OpExit), + } + return &sbpf.Program{Text: text} +} + +func runSha256Loop(p *sbpf.Program, fn sha256Call) (uint64, uint64, error) { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(10000000)} + vm := sbpf.NewInterpreter(p, &sbpf.VMOpts{Context: ctx, ComputeMeter: &ctx.ComputeMeter, Syscalls: func(hash uint32) (sbpf.Syscall, bool) { return sbpf.SyscallFunc3(fn), hash == hash_sol_sha256 }}) + ret, _, err := vm.Run() + used := ctx.ComputeMeter.Used() + vm.Finish() + return ret, used, err +} + +func TestSha256CapturedLoop(t *testing.T) { + const iterations = 1000 + p := sha256LoopProgram(iterations) + require.NoError(t, p.Verify()) + var data [36]byte + for i := uint32(0); i < iterations; i++ { + binary.LittleEndian.PutUint32(data[32:], i) + d := sha256.Sum256(data[:]) + copy(data[:32], d[:]) + } + want := binary.LittleEndian.Uint64(data[:8]) + var wantCU uint64 + for _, fn := range []sha256Call{sha256BaselineReference, SyscallSha256Impl} { + got, used, err := runSha256Loop(p, fn) + require.NoError(t, err) + require.Equal(t, want, got) + if wantCU == 0 { + wantCU = used + } + require.Equal(t, wantCU, used) + } +} + +func BenchmarkSha256CapturedLoop(b *testing.B) { + const iterations = 1000 + p := sha256LoopProgram(iterations) + for _, v := range []struct { + name string + fn sha256Call + }{ + {"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, + // Diagnostic lower bound: no hashing, translations or syscall CU charging. + // It is not a valid implementation and must never be used in replay. + {"dispatch_only", func(sbpf.VM, uint64, uint64, uint64) (uint64, error) { return 0, nil }}, + } { + b.Run(v.name, func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + if _, _, err := runSha256Loop(p, v.fn); err != nil { + b.Fatal(err) + } + } + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N*iterations), "ns/hash") + }) + } + b.Run("raw_chain", func(b *testing.B) { + var data [36]byte + b.ReportAllocs() + for j := 0; j < b.N; j++ { + clear(data[:]) + for i := uint32(0); i < iterations; i++ { + binary.LittleEndian.PutUint32(data[32:], i) + d := sha256.Sum256(data[:]) + copy(data[:32], d[:]) + } + } + sha256BenchDigest = sha256.Sum256(data[:]) + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N*iterations), "ns/hash") + }) +} diff --git a/pkg/sealevel/syscalls_sibling_pool_test.go b/pkg/sealevel/syscalls_sibling_pool_test.go new file mode 100644 index 000000000..a3b3b9e22 --- /dev/null +++ b/pkg/sealevel/syscalls_sibling_pool_test.go @@ -0,0 +1,42 @@ +package sealevel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +func TestSiblingHeaderWriteIsClearedOnFinish(t *testing.T) { + old := sbpf.UsePool + sbpf.UsePool = true + t.Cleanup(func() { sbpf.UsePool = old }) + for _, addr := range []uint64{sbpf.VaddrStack + 128, sbpf.VaddrHeap + 128} { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(100000), TransactionContext: &TransactionCtx{ + InstructionTrace: []InstructionCtx{{Data: []byte{1, 2, 3}}, {}, {}}, InstructionStack: []uint64{1}, + }} + vm := sbpf.NewInterpreter(&sbpf.Program{TextVA: sbpf.VaddrProgram}, &sbpf.VMOpts{HeapMax: 32768, Context: ctx, ComputeMeter: &ctx.ComputeMeter}) + // Keep a read-only view so the test itself does not mark the header dirty. + header, err := vm.Translate(addr, ProcessedSiblingInstructionSize, false) + require.NoError(t, err) + require.Equal(t, make([]byte, 16), header) + found, err := SyscallGetProcessedSiblingInstructionImpl(vm, 0, addr, 0, 0, 0) + require.NoError(t, err) + require.Equal(t, uint64(1), found) + require.Equal(t, byte(3), header[0]) + vm.Finish() + // Inspect before allocating another VM: this is deterministic and does not + // depend on sync.Pool choosing a particular backing buffer on the next Get. + require.Equal(t, make([]byte, 16), header) + } +} + +func TestSiblingHeaderRejectsReadOnlyDestination(t *testing.T) { + data := make([]byte, 16) + vm, ctx := newMemSyscallVM(t, nil, []sbpf.InputRegion{{RegionSize: 16, AddressSpaceReserved: 16, Data: data}}) + ctx.TransactionContext = &TransactionCtx{InstructionTrace: []InstructionCtx{{Data: []byte{1, 2, 3}}, {}, {}}, InstructionStack: []uint64{1}} + _, err := SyscallGetProcessedSiblingInstructionImpl(vm, 0, sbpf.VaddrInput, 0, 0, 0) + require.Error(t, err) + require.Equal(t, make([]byte, 16), data) +} diff --git a/pkg/sealevel/syscalls_sysvar.go b/pkg/sealevel/syscalls_sysvar.go index acd8929b4..400744653 100644 --- a/pkg/sealevel/syscalls_sysvar.go +++ b/pkg/sealevel/syscalls_sysvar.go @@ -12,7 +12,6 @@ import ( //"github.com/Overclock-Validator/mithril/pkg/mlog" "github.com/Overclock-Validator/mithril/pkg/safemath" "github.com/Overclock-Validator/mithril/pkg/sbpf" - "github.com/Overclock-Validator/mithril/pkg/util" "github.com/gagliardetto/solana-go" ) @@ -172,9 +171,9 @@ func SyscallGetEpochRewardsSysvarImpl(vm sbpf.VM, addr uint64) (uint64, error) { binary.LittleEndian.PutUint64(epochRewardsDst[8:16], epochRewards.NumPartitions) copy(epochRewardsDst[16:48], epochRewards.ParentBlockhash[:]) + // repr(C) u128 uses little-endian low/high limbs on the SBF target. binary.LittleEndian.PutUint64(epochRewardsDst[48:56], epochRewards.TotalPoints.Lo) binary.LittleEndian.PutUint64(epochRewardsDst[56:64], epochRewards.TotalPoints.Hi) - util.ReverseBytesInPlace(epochRewardsDst[48:64]) binary.LittleEndian.PutUint64(epochRewardsDst[64:72], epochRewards.TotalRewards) binary.LittleEndian.PutUint64(epochRewardsDst[72:80], epochRewards.DistributedRewards) diff --git a/pkg/sealevel/syscalls_sysvar_layout_test.go b/pkg/sealevel/syscalls_sysvar_layout_test.go new file mode 100644 index 000000000..1cea7ecb9 --- /dev/null +++ b/pkg/sealevel/syscalls_sysvar_layout_test.go @@ -0,0 +1,68 @@ +package sealevel + +import ( + "encoding/binary" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" + "math" + "testing" +) + +// Replace the old sysvars.so fixture, which expected packed EpochSchedule +// u64 fields at offsets 17/25. The syscall ABI is repr(C): offsets 24/32. +func TestSyscallSysvarLayouts(t *testing.T) { + vm, ctx := newMemSyscallVM(t, make([]byte, 512), nil) + ctx.Accounts = accounts.NewMemAccounts() + clock := SysvarClock{Slot: 1234, EpochStartTimestamp: 2222, Epoch: 1111, LeaderScheduleEpoch: 100000, UnixTimestamp: 3} + rent := SysvarRent{LamportsPerUint8Year: 12, ExemptionThreshold: 34, BurnPercent: 56} + schedule := SysvarEpochSchedule{SlotsPerEpoch: 1111, LeaderScheduleSlotOffset: 2222, Warmup: true, FirstNormalEpoch: 4444, FirstNormalSlot: 5555} + rewards := SysvarEpochRewards{DistributionStartingBlockHeight: 1234, NumPartitions: 4321, TotalRewards: 5656, DistributedRewards: 6767, Active: true} + rewards.TotalPoints.Lo = 0x0123456789abcdef + rewards.TotalPoints.Hi = 0xfedcba9876543210 + rewards.ParentBlockhash[0] = 0x73 + for _, key := range [][32]byte{SysvarClockAddr, SysvarRentAddr, SysvarEpochScheduleAddr, SysvarEpochRewardsAddr, SysvarLastRestartSlotAddr} { + require.NoError(t, ctx.Accounts.SetAccount(&key, &accounts.Account{Key: key, Lamports: 1})) + } + WriteClockSysvar(&ctx.Accounts, clock) + WriteRentSysvar(&ctx.Accounts, rent) + WriteEpochScheduleSysvar(&ctx.Accounts, schedule) + WriteEpochRewardsSysvar(&ctx.Accounts, rewards) + WriteLastRestartSlotSysvar(&ctx.Accounts, SysvarLastRestartSlot{LastRestartSlot: 989898}) + for _, tc := range []struct { + name string + call func(sbpf.VM, uint64) (uint64, error) + want []byte + }{ + {"clock", SyscallGetClockSysvarImpl, appendU64s(1234, 2222, 1111, 100000, 3)}, + {"rent", SyscallGetRentSysvarImpl, append(appendU64s(12, math.Float64bits(34)), 56, 0, 0, 0, 0, 0, 0, 0)}, + {"schedule", SyscallGetEpochScheduleSysvarImpl, appendU64s(1111, 2222, 1, 4444, 5555)}, + {"last restart", SyscallGetLastRestartSlotSysvarImpl, appendU64s(989898)}, + } { + t.Run(tc.name, func(t *testing.T) { + _, err := tc.call(vm, sbpf.VaddrInput) + require.NoError(t, err) + got, err := vm.Translate(sbpf.VaddrInput, uint64(len(tc.want)), false) + require.NoError(t, err) + require.Equal(t, tc.want, got) + clear(got) + }) + } + _, err := SyscallGetEpochRewardsSysvarImpl(vm, sbpf.VaddrInput) + require.NoError(t, err) + got, err := vm.Translate(sbpf.VaddrInput, 96, false) + require.NoError(t, err) + want := make([]byte, 96) + copy(want, appendU64s(1234, 4321)) + copy(want[16:48], rewards.ParentBlockhash[:]) + copy(want[48:], appendU64s(rewards.TotalPoints.Lo, rewards.TotalPoints.Hi, 5656, 6767)) + want[80] = 1 + require.Equal(t, want, got) +} +func appendU64s(values ...uint64) []byte { + var b []byte + for _, v := range values { + b = binary.LittleEndian.AppendUint64(b, v) + } + return b +} diff --git a/pkg/sealevel/sysvar_instructions_test.go b/pkg/sealevel/sysvar_instructions_test.go index 568fa8d98..67fb040cd 100644 --- a/pkg/sealevel/sysvar_instructions_test.go +++ b/pkg/sealevel/sysvar_instructions_test.go @@ -113,7 +113,7 @@ func TestExecute_Tx_Sysvar_Instructions_Bpf_Test(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) diff --git a/pkg/sealevel/types.go b/pkg/sealevel/types.go index 558429374..6d00a25f7 100644 --- a/pkg/sealevel/types.go +++ b/pkg/sealevel/types.go @@ -208,12 +208,12 @@ func (accountMeta *SolAccountMetaC) Marshal() ([]byte, error) { return nil, err } - err = binary.Write(buf, binary.LittleEndian, accountMeta.IsSigner) + err = binary.Write(buf, binary.LittleEndian, accountMeta.IsWritable) if err != nil { return nil, err } - err = binary.Write(buf, binary.LittleEndian, accountMeta.IsWritable) + err = binary.Write(buf, binary.LittleEndian, accountMeta.IsSigner) if err != nil { return nil, err } diff --git a/pkg/sealevel/vote_deque_ownership_test.go b/pkg/sealevel/vote_deque_ownership_test.go new file mode 100644 index 000000000..b7d50f43b --- /dev/null +++ b/pkg/sealevel/vote_deque_ownership_test.go @@ -0,0 +1,29 @@ +package sealevel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/gagliardetto/solana-go" + "github.com/gammazero/deque" + "github.com/stretchr/testify/require" +) + +func TestProcessNewVoteStateOwnsRetainedDeque(t *testing.T) { + // Model the TowerSync scratch deque being returned to its pool and reused. + scratch := new(deque.Deque[LandedVote]) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 100, ConfirmationCount: 2}}) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 101, ConfirmationCount: 1}}) + state := new(VoteState) + require.NoError(t, processNewVoteState(state, scratch, nil, nil, 0, 101, features.Features{})) + cached := newVoteState4FromCurrent(state, solana.PublicKey{}) + want := []LandedVote{state.Votes.At(0), state.Votes.At(1)} + + scratch.Clear() + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 200, ConfirmationCount: 2}}) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 201, ConfirmationCount: 1}}) + for i, vote := range want { + require.Equal(t, vote, state.Votes.At(i)) + require.Equal(t, vote, cached.Votes.At(i)) + } +} diff --git a/pkg/sealevel/vote_program.go b/pkg/sealevel/vote_program.go index 43a8500c2..679d1787d 100644 --- a/pkg/sealevel/vote_program.go +++ b/pkg/sealevel/vote_program.go @@ -1951,7 +1951,14 @@ func processNewVoteState(voteState *VoteState, newState *deque.Deque[LandedVote] } voteState.RootSlot = newRoot - voteState.Votes = *newState + // newState may be a pooled deque. Own the backing storage before its + // caller returns it to the pool: the resulting state can escape into the + // shared vote cache after this instruction completes. + var owned deque.Deque[LandedVote] + for i := 0; i < newState.Len(); i++ { + owned.PushBack(newState.At(i)) + } + voteState.Votes = owned return nil } diff --git a/pkg/sigverify/config_policy_test.go b/pkg/sigverify/config_policy_test.go new file mode 100644 index 000000000..52e8948ce --- /dev/null +++ b/pkg/sigverify/config_policy_test.go @@ -0,0 +1,63 @@ +package sigverify + +import ( + "runtime" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestResolveConfigPolicyDefaultsAndOverrides(t *testing.T) { + zero, err := ResolveConfig(Config{}) + require.NoError(t, err) + require.Equal(t, BackendAuto, zero.Backend) + require.Equal(t, min(2, runtime.GOMAXPROCS(0)), zero.Workers) + require.Equal(t, 8, zero.BatchTarget) + require.False(t, zero.DisableShredOverlap) + defaults, err := ResolveConfig(Defaults()) + require.NoError(t, err) + require.Equal(t, zero, defaults) + + override := Config{Backend: BackendGeneric, Workers: 4, BatchTarget: 4, DisableShredOverlap: true} + resolved, err := ResolveConfig(override) + require.NoError(t, err) + require.Equal(t, override, resolved) +} + +func TestInvalidPolicyDoesNotLatchBackend(t *testing.T) { + if !inChild() { + out, err := runConfigureChild(t, t.Name()) + require.NoError(t, err, "child output:\n%s", out) + require.Contains(t, out, "PASS") + return + } + before := Cfg + for _, cfg := range []Config{ + {Backend: BackendGeneric, Workers: -1}, + {Backend: BackendGeneric, BatchTarget: -1}, + {Backend: BackendGeneric, BatchTarget: 3}, + {Backend: BackendGeneric, BatchTarget: 16}, + } { + _, err := Configure(cfg) + require.Error(t, err) + require.Equal(t, before, Cfg, "invalid policy must not become live") + require.Empty(t, configuredBackend, "invalid policy must not latch the backend") + } + _, err := Configure(Config{Backend: BackendGeneric, Workers: 2, BatchTarget: 4}) + require.NoError(t, err, "valid configuration must remain possible after rejection") + require.Equal(t, 2, TransactionWorkers()) + require.Equal(t, 4, TransactionBatchTarget()) +} + +func TestTransactionPolicyWithoutConfigure(t *testing.T) { + if !inChild() { + out, err := runConfigureChild(t, t.Name()) + require.NoError(t, err, "child output:\n%s", out) + require.Contains(t, out, "PASS") + return + } + Cfg = Config{} + require.Equal(t, min(2, runtime.GOMAXPROCS(0)), TransactionWorkers()) + require.Equal(t, 8, TransactionBatchTarget()) + require.False(t, Cfg.DisableShredOverlap) +} diff --git a/pkg/sigverify/sigverify.go b/pkg/sigverify/sigverify.go index fae00447d..a0502712c 100644 --- a/pkg/sigverify/sigverify.go +++ b/pkg/sigverify/sigverify.go @@ -20,6 +20,7 @@ package sigverify import ( "fmt" + "runtime" "sync" narya "github.com/Overclock-Validator/narya-ed25519/ed25519" @@ -40,15 +41,60 @@ const ( BackendStdlib = "stdlib" ) -// Config selects the verification backend. It is deliberately tiny: the -// library's own defaults are good, and every knob here is a consensus-visible -// or performance-visible choice that an operator should have to state. +// Config selects the backend and the Turbine transaction-verification policy. +// Worker and batching settings do not change the TPU or fallback replay pools. type Config struct { - Backend string + Backend string + Workers int + BatchTarget int + DisableShredOverlap bool } // Defaults returns the configuration used when the operator sets nothing. -func Defaults() Config { return Config{Backend: BackendAuto} } +func Defaults() Config { return Config{Backend: BackendAuto, BatchTarget: BatchTarget} } + +// ResolveConfig validates every setting before the one-shot backend selection. +// Zero workers selects at most two transaction-verification workers; zero batch +// target uses eight signatures. Available short batches are never held to fill. +func ResolveConfig(cfg Config) (Config, error) { + if cfg.Backend == "" { + cfg.Backend = BackendAuto + } + switch cfg.Backend { + case BackendAuto, BackendR51, BackendGeneric, BackendStdlib: + default: + return Config{}, fmt.Errorf("sigverify.backend must be one of %q, %q, %q, %q; got %q", BackendAuto, BackendR51, BackendGeneric, BackendStdlib, cfg.Backend) + } + if cfg.Workers < 0 { + return Config{}, fmt.Errorf("sigverify.workers must be >= 0; got %d", cfg.Workers) + } + if cfg.Workers == 0 { + cfg.Workers = min(2, max(1, runtime.GOMAXPROCS(0))) + } + if cfg.BatchTarget == 0 { + cfg.BatchTarget = BatchTarget + } + if cfg.BatchTarget != 4 && cfg.BatchTarget != 8 { + return Config{}, fmt.Errorf("sigverify.batch_target must be 4 or 8 (0 uses 8); got %d", cfg.BatchTarget) + } + return cfg, nil +} + +// TransactionWorkers also supports callers that do not run node Configure. +func TransactionWorkers() int { + if Cfg.Workers > 0 { + return Cfg.Workers + } + return min(2, max(1, runtime.GOMAXPROCS(0))) +} + +// TransactionBatchTarget also supports the zero configuration outside startup. +func TransactionBatchTarget() int { + if Cfg.BatchTarget == 4 { + return 4 + } + return BatchTarget +} // Cfg is the live configuration, set once by Configure during startup and // read-only afterwards. It follows the same shape as replay.TrailingVerifierCfg. @@ -64,10 +110,6 @@ var Cfg = Defaults() // underlying library pins its backend on first use and a late switch would // leave the process in a state neither caller asked for. func Configure(cfg Config) (string, error) { - if cfg.Backend == "" { - cfg.Backend = Defaults().Backend - } - configureMu.Lock() defer configureMu.Unlock() @@ -79,12 +121,10 @@ func Configure(cfg Config) (string, error) { // Validate before publishing anything. Assigning Cfg first would leave a // rejected backend name visible to Backend() and to the startup log. - switch cfg.Backend { - case BackendAuto, BackendR51, BackendGeneric, BackendStdlib: - default: - return "", fmt.Errorf( - "sigverify.backend must be one of %q, %q, %q, %q; got %q", - BackendAuto, BackendR51, BackendGeneric, BackendStdlib, cfg.Backend) + var err error + cfg, err = ResolveConfig(cfg) + if err != nil { + return "", err } resolved, err := installBackend(cfg.Backend) diff --git a/pkg/statsd/statsd.go b/pkg/statsd/statsd.go index bbf3eea9c..5a9eda429 100644 --- a/pkg/statsd/statsd.go +++ b/pkg/statsd/statsd.go @@ -111,6 +111,7 @@ var ( TxsPerBlock = Metric{"txs_per_block"} SnapshotTarBytesRead = Metric{"snapshot_tar_bytes_read"} SlotReplays = Metric{"slot_replays"} + BlockProductionEntrySerializationErrors = Metric{"block_production_entry_serialization_errors_total"} BlockProductionLeaderSlots = Metric{"block_production_leader_slots_total"} BlockProductionLeaderSlotTerminals = Metric{"block_production_leader_slot_terminals_total"} BlockProductionParentReady = Metric{"block_production_parent_ready_activations_total"} @@ -124,6 +125,15 @@ var ( TurbineBlockDecode = Metric{"turbine_block_decode_duration_seconds"} TurbineTransactionParse = Metric{"turbine_transaction_parse_duration_seconds"} TurbineTransactionSigverify = Metric{"turbine_transaction_sigverify_duration_seconds"} + // Early durations sum elapsed component work, including verifier queueing; + // they overlap shred collection and are neither CPU nor pipeline wall time. + TurbineEarlyTransactionParse = Metric{"turbine_early_transaction_parse_duration_seconds"} + TurbineEarlyTransactionSigverify = Metric{"turbine_early_transaction_sigverify_elapsed_seconds"} + TurbineEarlyPreparationWait = Metric{"turbine_early_preparation_wait_seconds"} + TurbineEarlyVerifiedTransactions = Metric{"turbine_early_verified_transactions_total"} + TurbineFullToReady = Metric{"turbine_full_to_ready_duration_seconds"} + // ReplayFullToReplayed: last shred assembled -> replay result handed to consensus. + ReplayFullToReplayed = Metric{"replay_full_to_replayed_duration_seconds"} // ReplaySigverifyGroup times one drained group of transaction signatures // and ReplaySigverifyGroupSignatures counts how many signatures were in it. // The pair is what tells an operator whether batching is actually happening: @@ -229,11 +239,12 @@ var MetricToType = map[Metric]metricType{ SlotReplayDurationMs: TimingT, TxsPerBlock: TimingT, - SnapshotTarBytesRead: CountT, - SlotReplays: CountT, - BlockProductionLeaderSlots: CountT, - BlockProductionLeaderSlotTerminals: CountT, - BlockProductionParentReady: CountT, + SnapshotTarBytesRead: CountT, + SlotReplays: CountT, + BlockProductionEntrySerializationErrors: CountT, + BlockProductionLeaderSlots: CountT, + BlockProductionLeaderSlotTerminals: CountT, + BlockProductionParentReady: CountT, BlockProductionParentReadyAge: TimingT, BlockProductionStartCutoffLate: TimingT, @@ -245,6 +256,12 @@ var MetricToType = map[Metric]metricType{ TurbineBlockDecode: TimingT, TurbineTransactionParse: TimingT, TurbineTransactionSigverify: TimingT, + TurbineEarlyTransactionParse: TimingT, + TurbineEarlyTransactionSigverify: TimingT, + TurbineEarlyPreparationWait: TimingT, + TurbineEarlyVerifiedTransactions: CountT, + TurbineFullToReady: TimingT, + ReplayFullToReplayed: TimingT, ReplaySigverifyGroup: TimingT, ReplaySigverifyGroupSignatures: CountT, TurbineReplayAdmission: TimingT, @@ -335,13 +352,14 @@ var MetricToLabels = map[Metric][]string{ TasksIndexEntryBuilderLatency: {}, TasksAppendVecCopyingLatency: {}, - SlotReplayDurationMs: {}, - TxsPerBlock: {}, - SnapshotTarBytesRead: {}, - SlotReplays: {}, - BlockProductionLeaderSlots: {"outcome", "reason"}, - BlockProductionLeaderSlotTerminals: {"outcome", "terminal", "cause"}, - BlockProductionParentReady: {"activation", "status"}, + SlotReplayDurationMs: {}, + TxsPerBlock: {}, + SnapshotTarBytesRead: {}, + SlotReplays: {}, + BlockProductionEntrySerializationErrors: {}, + BlockProductionLeaderSlots: {"outcome", "reason"}, + BlockProductionLeaderSlotTerminals: {"outcome", "terminal", "cause"}, + BlockProductionParentReady: {"activation", "status"}, BlockProductionParentReadyAge: {"activation"}, BlockProductionStartCutoffLate: {"phase"}, @@ -353,6 +371,12 @@ var MetricToLabels = map[Metric][]string{ TurbineBlockDecode: {}, TurbineTransactionParse: {}, TurbineTransactionSigverify: {}, + TurbineEarlyTransactionParse: {}, + TurbineEarlyTransactionSigverify: {}, + TurbineEarlyPreparationWait: {}, + TurbineEarlyVerifiedTransactions: {}, + TurbineFullToReady: {}, + ReplayFullToReplayed: {}, ReplaySigverifyGroup: {}, ReplaySigverifyGroupSignatures: {}, TurbineReplayAdmission: {}, @@ -390,6 +414,11 @@ var MetricToBuckets = map[Metric][]float64{ TurbineBlockDecode: turbinePipelineDurationBuckets, TurbineTransactionParse: turbinePipelineDurationBuckets, TurbineTransactionSigverify: turbinePipelineDurationBuckets, + TurbineEarlyTransactionParse: turbinePipelineDurationBuckets, + TurbineEarlyTransactionSigverify: turbinePipelineDurationBuckets, + TurbineEarlyPreparationWait: turbinePipelineDurationBuckets, + TurbineFullToReady: turbinePipelineDurationBuckets, + ReplayFullToReplayed: turbinePipelineDurationBuckets, ReplaySigverifyGroup: turbinePipelineDurationBuckets, TurbineReplayAdmission: turbinePipelineDurationBuckets, AlpenglowVoteRewards: turbinePipelineDurationBuckets, diff --git a/pkg/statsd/statsd_test.go b/pkg/statsd/statsd_test.go index 830ea99eb..6ec3cd0c5 100644 --- a/pkg/statsd/statsd_test.go +++ b/pkg/statsd/statsd_test.go @@ -203,6 +203,10 @@ func TestTurbinePipelineDurationMetricsUseSecondsAndBoundedSchema(t *testing.T) TurbineBlockDecode, TurbineTransactionParse, TurbineTransactionSigverify, + TurbineEarlyTransactionParse, + TurbineEarlyTransactionSigverify, + TurbineEarlyPreparationWait, + TurbineFullToReady, TurbineReplayAdmission, } duration := 25 * time.Millisecond @@ -228,6 +232,11 @@ func TestTurbinePipelineDurationMetricsUseSecondsAndBoundedSchema(t *testing.T) } } +func TestTurbineEarlyVerifiedTransactionsHasBoundedCountSchema(t *testing.T) { + assert.Equal(t, CountT, MetricToType[TurbineEarlyVerifiedTransactions]) + assert.Equal(t, []string{}, MetricToLabels[TurbineEarlyVerifiedTransactions]) +} + func TestBlockProductionMetricLabelsStayBounded(t *testing.T) { assert.Equal(t, []string{"outcome", "reason"}, MetricToLabels[BlockProductionLeaderSlots]) assert.Equal(t, []string{"outcome", "terminal", "cause"}, MetricToLabels[BlockProductionLeaderSlotTerminals]) diff --git a/pkg/tpu/txfixture/readonly_pair.go b/pkg/tpu/txfixture/readonly_pair.go new file mode 100644 index 000000000..c1de0a074 --- /dev/null +++ b/pkg/tpu/txfixture/readonly_pair.go @@ -0,0 +1,42 @@ +package txfixture + +import ( + "crypto/ed25519" + "fmt" + + "github.com/gagliardetto/solana-go" +) + +const ReadonlyPairPoolSize = 128 +const ReadonlyPairCapacity = ReadonlyPairPoolSize * (ReadonlyPairPoolSize - 1) + +// ReadonlyPairWire builds the 198-byte, single-signature, zero-instruction +// workload used for leader packing tests. Varying ordered pairs of existing +// readonly accounts gives distinct messages without adding instructions or +// forcing a lookup of a new nonexistent account for every transaction. +// Reusing an ordinal requires a different payer or recent blockhash. +func ReadonlyPairWire(key ed25519.PrivateKey, hash solana.Hash, pool []solana.PublicKey, ordinal int) ([]byte, error) { + if len(key) != ed25519.PrivateKeySize || len(pool) != ReadonlyPairPoolSize || ordinal < 0 || ordinal >= ReadonlyPairCapacity { + return nil, fmt.Errorf("invalid readonly-pair fixture key, pool or ordinal") + } + n := ordinal * 7919 % ReadonlyPairCapacity + a, b := n/(ReadonlyPairPoolSize-1), n%(ReadonlyPairPoolSize-1) + if b >= a { + b++ + } + payer := solana.PublicKeyFromBytes(key.Public().(ed25519.PublicKey)) + if payer == pool[a] || payer == pool[b] || pool[a] == pool[b] { + return nil, fmt.Errorf("readonly-pair accounts must be distinct") + } + message := make([]byte, 0, 133) + message = append(message, 1, 0, 2, 3) + message = append(message, payer[:]...) + message = append(message, pool[a][:]...) + message = append(message, pool[b][:]...) + message = append(message, hash[:]...) + message = append(message, 0) + wire := make([]byte, 0, 198) + wire = append(wire, 1) + wire = append(wire, ed25519.Sign(key, message)...) + return append(wire, message...), nil +} diff --git a/pkg/tpu/txfixture/readonly_pair_test.go b/pkg/tpu/txfixture/readonly_pair_test.go new file mode 100644 index 000000000..70efa16dd --- /dev/null +++ b/pkg/tpu/txfixture/readonly_pair_test.go @@ -0,0 +1,68 @@ +package txfixture + +import ( + "crypto/ed25519" + "crypto/sha256" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Reproduce the full preloaded four-slot experiment without RPC, funds or sends. +func TestReadonlyPair200KDistinctMessages(t *testing.T) { + keys := make([]ed25519.PrivateKey, 8) + for i := range keys { + seed := sha256.Sum256([]byte{byte(i), 73}) + keys[i] = ed25519.NewKeyFromSeed(seed[:]) + } + pool := make([]solana.PublicKey, ReadonlyPairPoolSize) + for i := range pool { + pool[i] = solana.PublicKey{byte(i + 1), 77} + } + seen := make(map[[32]byte]struct{}, 200000) + for i := 0; i < 200000; i++ { + hash, index := solana.Hash{1}, i + if i >= 120000 { + hash, index = solana.Hash{2}, i-120000 + } + wire, err := ReadonlyPairWire(keys[index%8], hash, pool, index/8) + require.NoError(t, err) + require.Len(t, wire, 198) + h := sha256.Sum256(wire[65:]) + if _, ok := seen[h]; ok { + t.Fatalf("duplicate message %d", i) + } + seen[h] = struct{}{} + if i%ReadonlyPairCapacity == 0 || i == 119999 || i == 120000 || i == 199999 { + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + require.Len(t, tx.Signatures, 1) + require.Empty(t, tx.Message.Instructions) + require.Equal(t, hash, tx.Message.RecentBlockhash) + require.True(t, ed25519.Verify(ed25519.PublicKey(tx.Message.AccountKeys[0][:]), wire[65:], wire[1:65])) + } + } +} + +func TestReadonlyPairRejectsInvalidInputs(t *testing.T) { + key := ed25519.PrivateKey(PayerPrivateKey()) + pool := make([]solana.PublicKey, ReadonlyPairPoolSize) + for i := range pool { + pool[i] = solana.PublicKey{byte(i + 1), 77} + } + for _, ordinal := range []int{-1, ReadonlyPairCapacity} { + _, err := ReadonlyPairWire(key, TestBlockhash(), pool, ordinal) + require.Error(t, err) + } + _, err := ReadonlyPairWire(nil, TestBlockhash(), pool, 0) + require.Error(t, err) + _, err = ReadonlyPairWire(key, TestBlockhash(), nil, 0) + require.Error(t, err) + pool[0] = solana.PublicKeyFromBytes(key.Public().(ed25519.PublicKey)) + _, err = ReadonlyPairWire(key, TestBlockhash(), pool, 0) + require.Error(t, err) + pool[0], pool[1] = solana.PublicKey{9}, solana.PublicKey{9} + _, err = ReadonlyPairWire(key, TestBlockhash(), pool, 0) + require.Error(t, err) +} diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 9b69f4370..6ef1d89d8 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -10,6 +10,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/statsd" + "github.com/Overclock-Validator/mithril/pkg/turbine/internal/rsrecover" "github.com/gagliardetto/solana-go" "github.com/klauspost/reedsolomon" ) @@ -43,20 +44,26 @@ const ( ) type SlotAssembler struct { - mu sync.Mutex - slots map[uint64]*slotState - completedSlots map[uint64]struct{} - knownBlockIDs map[uint64]solana.Hash - rejectedBlockIDs map[uint64]map[solana.Hash]struct{} - protectedKnownIDs map[uint64]struct{} - protectedBlockIDs map[uint64]struct{} - priorityRepairSlots map[uint64]struct{} - priorityRepairOrder []uint64 - encoders map[fecLayout]reedsolomon.Encoder - partialShredObs map[uint64]PartialShredObservation // shreds seen for slots that never became full (retained for skip observability) - retentionFloor uint64 // when non-zero, slots >= floor are never "too old" (repair catchup holds a window far behind the live edge) - edgeScanLag uint64 // how far behind the shred edge the freshness-repair scan reaches (0 = repairScanSlotWindow) - maxObservedSlot uint64 + mu sync.Mutex + slots map[uint64]*slotState + completedSlots map[uint64]struct{} + knownBlockIDs map[uint64]solana.Hash + rejectedBlockIDs map[uint64]map[solana.Hash]struct{} + protectedKnownIDs map[uint64]struct{} + protectedBlockIDs map[uint64]struct{} + priorityRepairSlots map[uint64]struct{} + priorityRepairOrder []uint64 + encoders map[fecLayout]reedsolomon.Encoder + partialShredObs map[uint64]PartialShredObservation // shreds seen for slots that never became full (retained for skip observability) + retentionFloor uint64 // when non-zero, slots >= floor are never "too old" (repair catchup holds a window far behind the live edge) + edgeScanLag uint64 // how far behind the shred edge the freshness-repair scan reaches (0 = repairScanSlotWindow) + maxObservedSlot uint64 + // Age sweeps depend on the edge, repair floor, and mutations that can add + // old metadata or release a completing generation's protected identities. + retentionSwept bool + retentionDirty bool + retentionSweepEdge uint64 + retentionSweepFloor uint64 highestFullSlot uint64 // monotonic: highest slot reconstructed from shreds ("full", Agave SlotMeta/is_full sense) recoveredDataShreds uint64 usefulRepairShreds uint64 // distinct data shreds delivered BY repair (the throughput signal) @@ -72,6 +79,16 @@ type SlotAssembler struct { // Configured before ingestion; tests may replace it with a blocking probe. // Production uses the process-wide bounded transaction verifier. verifyTransactions func(context.Context, *block.Block) error + entryPrefetch *entryPrefetchPool + // Streaming feed subscriber (see stream.go); nil when nothing consumes + // batches before completion. + streamSubscriber chan<- StreamEvent + streamDroppedEvents uint64 + streamRepairParent *slotState + streamRepairChild *slotState + streamRepairInvalidChild *slotState + streamRepairUntil time.Time + streamRepairWake chan<- struct{} } type SlotRepairRequest struct { @@ -91,14 +108,15 @@ type PartialShredObservation struct { } type slotState struct { - slot uint64 - parentSlot uint64 - shreds map[uint32]*Shred - fecSets map[uint32]*fecState - lastIndex uint32 - haveLast bool - shredVer uint16 - firstParent bool + pipelineTrace *entryPipelineTrace + slot uint64 + parentSlot uint64 + shreds map[uint32]*Shred + fecSets map[uint32]*fecState + lastIndex uint32 + haveLast bool + shredVer uint16 + firstParent bool // Observability: when the slot's first shred was accepted, and how many of // its shreds arrived via repair rather than turbine. @@ -106,12 +124,20 @@ type slotState struct { fullAt time.Time repairedShreds int // completing makes the immutable full state a single-owner generation token. - completing bool + completing bool + batchIndex *entryBatchIndex + completeBatches []shredBatchRange + prefetch *slotEntryPrefetch // Assembly failures for this slot (mixed variants/signatures, FEC layout // conflicts, ...). A slot frozen below completion while repair responses // flow is usually poisoned state — the latest error names the poison. errCount int lastErr string + // streamCompleted marks a generation whose complete block was accepted, so + // the feed's release event says "completed" rather than "cancelled"; + // streamCancelReason names the discard path otherwise. + streamCompleted bool + streamCancelReason string } func (s *slotState) noteError(err error) { @@ -120,6 +146,8 @@ func (s *slotState) noteError(err error) { } type slotCompletionWork struct { + // Captured under mu: cancelled prefetch readers are not completion inputs. + ignorePrefetch bool state *slotState queuedAt time.Time observeCollection bool @@ -138,10 +166,11 @@ type processedSlotCompletion struct { } type slotCompletionResult struct { - block *block.Block - err error - hydrated bool - pending bool + generation StreamGeneration + block *block.Block + err error + hydrated bool + pending bool } type fecLayout struct { @@ -161,6 +190,7 @@ type fecState struct { haveSig bool dataVariant byte codeVariant byte + rootCache *authenticatedFECRoot } func NewSlotAssembler() *SlotAssembler { @@ -174,7 +204,6 @@ func NewSlotAssembler() *SlotAssembler { priorityRepairSlots: make(map[uint64]struct{}), partialShredObs: make(map[uint64]PartialShredObservation), encoders: make(map[fecLayout]reedsolomon.Encoder), - verifyTransactions: validateBlockTransactionsContext, } } @@ -185,6 +214,7 @@ func (a *SlotAssembler) recordPartialObsLocked(state *slotState) { if state == nil || len(state.shreds) == 0 { return } + a.retentionDirty = true a.partialShredObs[state.slot] = PartialShredObservation{ DataShreds: len(state.shreds), RepairedShreds: state.repairedShreds, @@ -239,6 +269,17 @@ func (a *SlotAssembler) AddShredFrom(shred *Shred, fromRepair bool) (*block.Bloc // reconstructable it returns a single immutable completion token; decoding, // parsing, and signature verification must happen after this method unlocks. func (a *SlotAssembler) addShredFrom(shred *Shred, fromRepair bool) (*slotCompletionWork, error) { + return a.addShredFromWithRoot(shred, fromRepair, nil) +} + +// addShredFromWithRoot accepts an optional result from successful authentication +// of this immutable shred. Unauthenticated callers and spool hydration use nil +// and retain the normal root-computation fallback. +func (a *SlotAssembler) addShredFromWithRoot(shred *Shred, fromRepair bool, root *solana.Hash) (*slotCompletionWork, error) { + var admissionEntered int64 + if shred != nil && entryTraceSelected(shred.Slot) { + admissionEntered = entryTraceNow() + } if shred == nil { return nil, nil } @@ -262,6 +303,16 @@ func (a *SlotAssembler) addShredFrom(shred *Shred, fromRepair bool) (*slotComple a.ignoredOldShreds++ return nil, nil } + // Diagnostic only; the deferred observation runs while a.mu is still held. + var repairTrace *entryRepairTrace + if fromRepair && admissionEntered != 0 { + repairTrace = &entryRepairTrace{Event: "repair_admission", Origin: entryTraceOrigin.UnixNano(), Slot: shred.Slot, Index: shred.Index, FEC: shred.FECSetIndex, AdmissionStart: admissionEntered, DeficitBefore: traceFECDeficit(state, shred.FECSetIndex), Outcome: "rejected"} + defer func() { + repairTrace.ResponseAt = entryTraceNow() + repairTrace.DeficitAfter = traceFECDeficit(state, shred.FECSetIndex) + emitRepairTrace(*repairTrace) + }() + } var err error switch shred.Type { case ShredTypeData: @@ -282,14 +333,28 @@ func (a *SlotAssembler) addShredFrom(shred *Shred, fromRepair bool) (*slotComple } if err != nil { if errors.Is(err, ErrDuplicateShred) { + if repairTrace != nil { + repairTrace.Outcome = "duplicate" + } return nil, nil } state.noteError(err) return nil, err } + source := entryShredSource{Path: "non_repair", FEC: shred.FECSetIndex, TriggerIndex: shred.Index, TriggerCoding: shred.Type == ShredTypeCode, TriggerRepair: fromRepair, AdmissionEntered: admissionEntered} + if fromRepair { + source.Path = "repair" + } + state.traceAcceptedShred(shred, source) + a.notePrefetchShredLocked(state, shred) if state.firstShredAt.IsZero() { state.firstShredAt = time.Now() } + if root != nil { + if fec := state.fecSets[shred.FECSetIndex]; fec != nil { + fec.rememberAuthenticatedRoot(shred, *root) + } + } recovered, err := a.recoverFEC(state, shred.FECSetIndex) if err != nil { @@ -303,10 +368,18 @@ func (a *SlotAssembler) addShredFrom(shred *Shred, fromRepair bool) (*slotComple return nil, err } if err == nil { + source.Path = "fec_recovery" + state.traceAcceptedShred(recoveredShred, source) + a.notePrefetchShredLocked(state, recoveredShred) a.recoveredDataShreds++ } } + if repairTrace != nil { + repairTrace.Outcome = "accepted" + repairTrace.Recovered = len(recovered) + } + a.prefetchEntriesLocked(state) if !state.complete() { return nil, nil } @@ -318,6 +391,9 @@ func (a *SlotAssembler) claimCompletionLocked(state *slotState, reportNonCanonic return nil } now := time.Now() + if state.pipelineTrace != nil { + state.pipelineTrace.sealed = true + } observeCollection := state.fullAt.IsZero() if observeCollection { state.fullAt = now @@ -332,6 +408,7 @@ func (a *SlotAssembler) claimCompletionLocked(state *slotState, reportNonCanonic } return &slotCompletionWork{ state: state, + ignorePrefetch: state.prefetch != nil && state.prefetch.released, queuedAt: now, observeCollection: observeCollection, reportNonCanonical: reportNonCanonical, @@ -348,6 +425,7 @@ func (a *SlotAssembler) abortCompletion(work *slotCompletionWork) { } a.mu.Lock() if a.slots[work.state.slot] == work.state && work.state.completing { + a.retentionDirty = true work.state.completing = false } a.mu.Unlock() @@ -363,6 +441,7 @@ func (a *SlotAssembler) processCompletion(ctx context.Context, work *slotComplet if ctx.Err() != nil { return processedSlotCompletion{canceled: true} } + ctx = withEntryPipelineTrace(ctx, work.state.pipelineTrace) startedAt := time.Now() timings := block.TurbineIngressTimings{CompletionQueueDelay: startedAt.Sub(work.queuedAt)} _ = statsd.Duration(statsd.TurbineBlockCompletionQueueDelay, timings.CompletionQueueDelay, nil) @@ -374,19 +453,27 @@ func (a *SlotAssembler) processCompletion(ctx context.Context, work *slotComplet } decodeStartedAt := time.Now() - var decodeTimings entryDecodeTimings + decodeTimings := entryDecodeTimings{ctx: ctx} + if work.state.prefetch != nil && !work.ignorePrefetch { + decodeTimings.prefetched = work.state.prefetch.batches + } blk, parentInfo, roots, err := work.state.decodeBlock(&decodeTimings) decodeTotal := time.Since(decodeStartedAt) - decodeOnly := decodeTotal - decodeTimings.transactionParse + decodeOnly := decodeTotal - decodeTimings.transactionParse - decodeTimings.prefetchWait if decodeOnly < 0 { decodeOnly = 0 } timings.BlockDecode = decodeOnly timings.TransactionParse = decodeTimings.transactionParse + timings.EarlyPreparationWait = decodeTimings.prefetchWait _ = statsd.Duration(statsd.TurbineBlockDecode, timings.BlockDecode, nil) _ = statsd.Duration(statsd.TurbineTransactionParse, timings.TransactionParse, nil) processed := processedSlotCompletion{block: blk, parentInfo: parentInfo, roots: roots, err: err, timings: timings} if err != nil { + if ctx.Err() != nil { + processed.canceled = true + processed.err = nil + } return processed } if ctx.Err() != nil { @@ -395,7 +482,12 @@ func (a *SlotAssembler) processCompletion(ctx context.Context, work *slotComplet } sigverifyStartedAt := time.Now() - processed.err = work.verifyTransactions(ctx, blk) + if work.state.prefetch != nil && len(decodeTimings.retained) > 0 { + processed.err = verifyDecodedEntryBatchesWithTimings(ctx, blk, decodeTimings.retained, work.state.prefetch.pool.verifier, &decodeTimings) + } else { + processed.err = work.verifyTransactions(ctx, blk) + } + earlyEntryTimings(&decodeTimings, work.state.fullAt, &processed.timings) processed.timings.TransactionSigverify = time.Since(sigverifyStartedAt) _ = statsd.Duration(statsd.TurbineTransactionSigverify, processed.timings.TransactionSigverify, nil) if ctx.Err() != nil { @@ -406,6 +498,13 @@ func (a *SlotAssembler) processCompletion(ctx context.Context, work *slotComplet if processed.err == nil { blk.MarkTransactionSignaturesVerified() processed.completionReadyAt = time.Now() + processed.timings.FullToReady = processed.completionReadyAt.Sub(work.state.fullAt) + _ = statsd.Duration(statsd.TurbineFullToReady, processed.timings.FullToReady, nil) + _ = statsd.Duration(statsd.TurbineEarlyPreparationWait, processed.timings.EarlyPreparationWait, nil) + _ = statsd.Duration(statsd.TurbineEarlyTransactionParse, processed.timings.EarlyTransactionParse, nil) + _ = statsd.Duration(statsd.TurbineEarlyTransactionSigverify, processed.timings.EarlyTransactionSigverify, nil) + _ = statsd.Count(statsd.TurbineEarlyVerifiedTransactions, int64(processed.timings.EarlyVerifiedTransactions), nil) + queueEntryPipelineReport(work.state, blk, &decodeTimings, startedAt, processed.completionReadyAt) } return processed } @@ -420,11 +519,22 @@ func (a *SlotAssembler) finalizeCompletion(work *slotCompletionWork, processed p a.mu.Unlock() return nil, nil } + // Every terminal outcome releases this generation's retention protection. + a.retentionDirty = true + if processed.canceled { + state.completing = false + a.mu.Unlock() + return nil, nil + } if processed.err != nil { // Retain a deterministic decode/verification failure on the live full // state so catchup diagnostics report poison instead of a missing slot. state.noteError(processed.err) state.completing = false + // Diagnostics retain the poisoned slot, not a usable stream. Cancel + // readers now; cleanup returns capacity only after they have joined. + state.streamCancelReason = "completion_failed" + a.releasePrefetchLocked(state) a.mu.Unlock() return nil, processed.err } @@ -434,6 +544,8 @@ func (a *SlotAssembler) finalizeCompletion(work *slotCompletionWork, processed p if !a.acceptAlpenglowBlockIDLocked(blk) { a.trackNonCanonicalBlockIDLocked(blk) a.recordPartialObsLocked(state) + state.streamCancelReason = "non_canonical" + a.releasePrefetchLocked(state) delete(a.slots, state.slot) a.mu.Unlock() if work.reportNonCanonical { @@ -442,6 +554,8 @@ func (a *SlotAssembler) finalizeCompletion(work *slotCompletionWork, processed p return nil, nil } + state.streamCompleted = true + a.releasePrefetchLocked(state) delete(a.slots, state.slot) a.completedSlots[state.slot] = struct{}{} a.trackBlockIDLocked(blk) @@ -510,6 +624,9 @@ func (a *SlotAssembler) SetKnownAlpenglowBlockID(slot uint64, blockID solana.Has if _, rejected := a.rejectedBlockIDs[slot][blockID]; rejected { return } + if _, exists := a.knownBlockIDs[slot]; !exists { + a.retentionDirty = true + } a.knownBlockIDs[slot] = blockID } @@ -530,6 +647,7 @@ func (a *SlotAssembler) RejectAlpenglowBlockID(slot uint64, blockID solana.Hash) if ids == nil { ids = make(map[solana.Hash]struct{}) a.rejectedBlockIDs[slot] = ids + a.retentionDirty = true } ids[blockID] = struct{}{} if a.knownBlockIDs[slot] == blockID { @@ -540,8 +658,20 @@ func (a *SlotAssembler) RejectAlpenglowBlockID(slot uint64, blockID solana.Hash) func (a *SlotAssembler) ResetSlot(slot uint64) { a.mu.Lock() defer a.mu.Unlock() + if a.streamRepairParent != nil && a.streamRepairParent.slot == slot { + a.streamRepairParent = nil + a.streamRepairChild = nil + } + if a.streamRepairChild != nil && a.streamRepairChild.slot == slot { + a.streamRepairChild = nil + } + a.retentionDirty = true a.recordPartialObsLocked(a.slots[slot]) + if state := a.slots[slot]; state != nil { + state.streamCancelReason = "reset" + } + a.releasePrefetchLocked(a.slots[slot]) delete(a.slots, slot) delete(a.completedSlots, slot) } @@ -551,8 +681,13 @@ func (a *SlotAssembler) PrioritizeRepairSlot(slot uint64) { } func (a *SlotAssembler) PrioritizeRepairRange(start, end uint64) { + a.prioritizeRepairRange(start, end) +} + +// Report only newly installed pins, so repeated replay hints do not wake repair. +func (a *SlotAssembler) prioritizeRepairRange(start, end uint64) bool { if start == 0 { - return + return false } if end < start { end = start @@ -564,11 +699,13 @@ func (a *SlotAssembler) PrioritizeRepairRange(start, end uint64) { a.mu.Lock() defer a.mu.Unlock() + changed := false for slot := start; ; slot++ { if _, completed := a.completedSlots[slot]; !completed { if _, exists := a.priorityRepairSlots[slot]; !exists { a.priorityRepairSlots[slot] = struct{}{} a.priorityRepairOrder = append(a.priorityRepairOrder, slot) + changed = true } } if slot == end { @@ -576,6 +713,7 @@ func (a *SlotAssembler) PrioritizeRepairRange(start, end uint64) { } } a.prunePriorityRepairSlotsLocked() + return changed } func (a *SlotAssembler) slotState(slot uint64, version uint16) *slotState { @@ -635,6 +773,35 @@ func (a *SlotAssembler) slotTooOldLocked(slot uint64) bool { } func (a *SlotAssembler) pruneOldSlotsLocked() { + if !a.retentionSwept || a.retentionDirty || a.retentionSweepEdge != a.maxObservedSlot || a.retentionSweepFloor != a.retentionFloor { + a.sweepRetentionMapsLocked() + a.retentionSwept = true + a.retentionDirty = false + a.retentionSweepEdge = a.maxObservedSlot + a.retentionSweepFloor = a.retentionFloor + } + // New incomplete generations can exceed the cap without advancing the + // edge (especially during catch-up). Never cache the capacity check. + + if len(a.slots) == 0 { + return + } + for len(a.slots) > maxRetainedIncompleteSlotCap { + victim, ok := a.capEvictionCandidateLocked() + if !ok { + return + } + a.recordPartialObsLocked(a.slots[victim]) + if state := a.slots[victim]; state != nil { + state.streamCancelReason = "evicted" + } + a.releasePrefetchLocked(a.slots[victim]) + delete(a.slots, victim) + a.evictedSlots++ + } +} + +func (a *SlotAssembler) sweepRetentionMapsLocked() { if len(a.slots) > 0 && a.maxObservedSlot > maxRetainedIncompleteSlotLag { minSlot := a.maxObservedSlot - maxRetainedIncompleteSlotLag if a.retentionFloor > 0 && a.retentionFloor < minSlot { @@ -643,6 +810,8 @@ func (a *SlotAssembler) pruneOldSlotsLocked() { for slot, state := range a.slots { if slot < minSlot && !state.completing { a.recordPartialObsLocked(state) + state.streamCancelReason = "retention" + a.releasePrefetchLocked(a.slots[slot]) delete(a.slots, slot) a.evictedSlots++ } @@ -686,18 +855,6 @@ func (a *SlotAssembler) pruneOldSlotsLocked() { } a.prunePriorityRepairSlotsLocked() - if len(a.slots) == 0 { - return - } - for len(a.slots) > maxRetainedIncompleteSlotCap { - victim, ok := a.capEvictionCandidateLocked() - if !ok { - return - } - a.recordPartialObsLocked(a.slots[victim]) - delete(a.slots, victim) - a.evictedSlots++ - } } // capEvictionCandidateLocked chooses state furthest ahead of replay, rather @@ -809,6 +966,10 @@ func (s *slotState) addDataShred(shred *Shred) error { } func (s *slotState) repairRequest(maxMissing int) (SlotRepairRequest, bool) { + return s.repairRequestWithPrefix(maxMissing, false) +} + +func (s *slotState) repairRequestWithPrefix(maxMissing int, prefix bool) (SlotRepairRequest, bool) { req := SlotRepairRequest{Slot: s.slot} var maxObserved uint32 @@ -825,7 +986,7 @@ func (s *slotState) repairRequest(maxMissing int) (SlotRepairRequest, bool) { req.HighestDataShredIndex = maxObserved + 1 } - req.MissingDataShreds = s.missingDataForRepair(maxObserved, maxMissing) + req.MissingDataShreds = s.missingDataForRepairWithPrefix(maxObserved, maxMissing, prefix) if len(req.MissingDataShreds) == 0 && !req.NeedHighestDataShred { return SlotRepairRequest{}, false @@ -874,6 +1035,10 @@ func (span codedSpan) requestsToUnlock() int { // promises more — without that, the tail waits on a HighestWindowIndex // round trip to be discovered. func (s *slotState) missingDataForRepair(maxObserved uint32, maxMissing int) []uint32 { + return s.missingDataForRepairWithPrefix(maxObserved, maxMissing, false) +} + +func (s *slotState) missingDataForRepairWithPrefix(maxObserved uint32, maxMissing int, prefix bool) []uint32 { spans := make([]codedSpan, 0, len(s.fecSets)) for _, fec := range s.fecSets { if !fec.haveLayout || fec.layout.dataShreds == 0 { @@ -933,7 +1098,38 @@ func (s *slotState) missingDataForRepair(maxObserved uint32, maxMissing int) []u return spans[order[a]].start < spans[order[b]].start }) + // For a streaming head, one earliest hole gates every later batch. Move + // just that span ahead of cheapest-unlock order; retain deficit capping + // and every existing request/admission limit. Without a known layout, + // prioritize one earliest missing data index rather than guessing a span. + var first []uint32 + if prefix { + earliest := -1 + for _, i := range order { + if earliest < 0 || spans[i].missing[0] < spans[earliest].missing[0] { + earliest = i + } + } + if len(uncovered) > 0 && (earliest < 0 || uncovered[0] < spans[earliest].missing[0]) { + first = uncovered[:1] + uncovered = uncovered[1:] + } else if earliest >= 0 { + first = spans[earliest].missing[:spans[earliest].requestsToUnlock()] + for j, i := range order { + if i == earliest { + order = append(order[:j], order[j+1:]...) + break + } + } + } + } missing := make([]uint32, 0, min(maxMissing, 64)) + for _, index := range first { + if len(missing) >= maxMissing { + return missing + } + missing = append(missing, index) + } for _, i := range order { span := spans[i] for _, index := range span.missing[:span.requestsToUnlock()] { @@ -1018,6 +1214,9 @@ func (a *SlotAssembler) trackBlockIDLocked(blk *block.Block) { if known, ok := a.knownBlockIDs[blk.Slot]; ok && known != (solana.Hash{}) && known != blockID { return } + if _, exists := a.knownBlockIDs[blk.Slot]; !exists { + a.retentionDirty = true + } a.knownBlockIDs[blk.Slot] = blockID } @@ -1219,7 +1418,7 @@ func (a *SlotAssembler) RepairRequestsTiered(maxSlots int, maxMissingPerSlot int HighestDataShredIndex: 0, }) } - if req, ok := state.repairRequest(maxMissing); ok { + if req, ok := state.repairRequestWithPrefix(maxMissing, priorityPin && len(dst) == 0 && a.streamSubscriber != nil); ok { seen[slot] = struct{}{} return append(dst, req) } @@ -1227,7 +1426,11 @@ func (a *SlotAssembler) RepairRequestsTiered(maxSlots int, maxMissingPerSlot int } a.prunePriorityRepairSlotsLocked() - for _, slot := range a.priorityRepairOrder { + // Pin insertion order is retention policy, not dependency order. An older + // parent can be discovered after its child; give that parent the head share. + ordered := append([]uint64(nil), a.priorityRepairOrder...) + sort.Slice(ordered, func(i, j int) bool { return ordered[i] < ordered[j] }) + for _, slot := range ordered { // HEAD FIRST: the first priority slot — the one gating emission — // may list up to repairHeadMaxMissing, several times the per-slot // cap, so its admission share stays full at any response latency. @@ -1300,21 +1503,48 @@ func (a *SlotAssembler) recoverFEC(state *slotState, fecSetIndex uint32) ([]*Shr } shards[int(layout.dataShreds)+int(pos)] = shard } - encoder, err := a.fecEncoder(layout) - if err != nil { - return nil, err - } - required := make([]bool, int(layout.dataShreds)+int(layout.codingShreds)) var missingData int + missingDataIndex := -1 for idx := 0; idx < int(layout.dataShreds); idx++ { if fec.data[uint32(idx)] == nil { - required[idx] = true missingData++ + missingDataIndex = idx } } if missingData == 0 { return nil, nil } + if missingData == 1 && + layout.dataShreds == rsrecover.DataShards && + layout.codingShreds == rsrecover.CodingShards { + presence, err := rsrecover.Presence(shards) + if err != nil { + return nil, err + } + dst := make([]byte, layout.shardSize) + if err := rsrecover.RecoverOneData(presence, missingDataIndex, shards, dst); err != nil { + return nil, fmt.Errorf("recover one FEC data shred slot %d fec_set=%d: %w", state.slot, fecSetIndex, err) + } + shred, err := fec.recoveredDataShred(uint32(missingDataIndex), dst) + if err != nil { + return nil, err + } + shards[missingDataIndex] = dst + recovered := []*Shred{shred} + if err := a.authenticateRecoveredFEC(fec, shards, recovered); err != nil { + return nil, err + } + return recovered, nil + } + + required := make([]bool, int(layout.dataShreds)+int(layout.codingShreds)) + for idx := 0; idx < int(layout.dataShreds); idx++ { + required[idx] = fec.data[uint32(idx)] == nil + } + encoder, err := a.fecEncoder(layout) + if err != nil { + return nil, err + } if err := encoder.ReconstructSome(shards, required); err != nil { if errors.Is(err, reedsolomon.ErrTooFewShards) { return nil, nil @@ -1337,6 +1567,9 @@ func (a *SlotAssembler) recoverFEC(state *slotState, fecSetIndex uint32) ([]*Shr } recovered = append(recovered, shred) } + if err := a.authenticateRecoveredFEC(fec, shards, recovered); err != nil { + return nil, err + } return recovered, nil } @@ -1498,6 +1731,24 @@ func (s *slotState) complete() bool { } func (s *slotState) orderedShreds() []*Shred { + // Normal completed slots contain exactly the contiguous range 0..lastIndex. + // Keep the sparse fallback: malformed tails and focused partial-state callers + // must not silently lose shreds beyond the last-in-slot marker. + if s.haveLast && uint64(len(s.shreds)) == uint64(s.lastIndex)+1 { + out := make([]*Shred, len(s.shreds)) + for idx := range out { + shred := s.shreds[uint32(idx)] + if shred == nil || shred.Index != uint32(idx) { + return s.sortedShreds() + } + out[idx] = shred + } + return out + } + return s.sortedShreds() +} + +func (s *slotState) sortedShreds() []*Shred { indexes := make([]int, 0, len(s.shreds)) for idx := range s.shreds { indexes = append(indexes, int(idx)) @@ -1514,7 +1765,7 @@ func (s *slotState) orderedShreds() []*Shred { // transaction signature verification. Parent/child identity hints are applied // later under the assembler lock so hints learned while this runs still win. func (s *slotState) decodeBlock(timings *entryDecodeTimings) (*block.Block, *AlpenglowParentInfo, []solana.Hash, error) { - entries, parentInfo, footer, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(s.orderedShreds(), timings) + entries, parentInfo, footer, err := decodeEntriesFromOrderedDataShreds(s.orderedShreds(), timings) if err != nil { return nil, nil, nil, err } @@ -1660,35 +1911,36 @@ func (s *slotState) fecSetMerkleRoots() ([]solana.Hash, error) { } func (f *fecState) merkleRoot() (solana.Hash, bool, error) { - for _, idx := range sortedUint32Keys(f.data) { - shred := f.data[idx] - if shred == nil || shred.Recovered { - continue - } - root, err := shred.MerkleRoot() - if err != nil { - if errors.Is(err, ErrUnsupportedShred) { - continue + // Preserve the old lowest-data-index, then lowest-coding-position choice, + // including its first error. Unsupported variants and recovered data have + // no usable proof. Selecting the minimum needs neither sorting nor scratch. + selected := f.data[0] + var first uint32 + // Relative index zero is already the minimum. Repair or legacy/malformed + // inputs may lack that proof and still take the general selection path. + if !hasMerkleRootProof(selected) || selected.Recovered { + selected = nil + for idx, shred := range f.data { + if hasMerkleRootProof(shred) && !shred.Recovered && (selected == nil || idx < first) { + selected, first = shred, idx } - return solana.Hash{}, false, err } - return root, true, nil } - for _, pos := range sortedUint16Keys(f.coding) { - shred := f.coding[pos] - if shred == nil { - continue - } - root, err := shred.MerkleRoot() - if err != nil { - if errors.Is(err, ErrUnsupportedShred) { - continue + if selected == nil { + for pos, shred := range f.coding { + if hasMerkleRootProof(shred) && (selected == nil || uint32(pos) < first) { + selected, first = shred, uint32(pos) } - return solana.Hash{}, false, err } - return root, true, nil } - return solana.Hash{}, false, nil + if selected == nil { + return solana.Hash{}, false, nil + } + if cached := f.rootCache; cached != nil && cached.matches(selected) { + return cached.root, true, nil + } + root, err := selected.MerkleRoot() + return root, err == nil, err } func merkleTreeRoot(leaves []solana.Hash) solana.Hash { diff --git a/pkg/turbine/assembler_test.go b/pkg/turbine/assembler_test.go index f70f3d695..e38b468b3 100644 --- a/pkg/turbine/assembler_test.go +++ b/pkg/turbine/assembler_test.go @@ -622,6 +622,15 @@ func TestDecodeAlpenglowParentMarkers(t *testing.T) { func TestSlotAssemblerRecoversMissingMerkleDataShredFromCodingShreds(t *testing.T) { dataShreds := localnetMerkleShreds(t, "d") codeShreds := localnetMerkleShreds(t, "c") + // These 2022 fixtures supply a useful non-power-of-two, unchained 1+17 + // erasure layout, but their old proofs do not yield a common root under + // today's Merkle hashing. Re-sign each complete tree using current proofs; + // recovery must now authenticate the tree, not just reconstruct the data. + for i, data := range dataShreds { + packets := append([][]byte{data}, codeShreds[i*17:(i+1)*17]...) + resignRecoveryFixture(t, packets) + } + if len(dataShreds) < 2 || len(codeShreds) == 0 { t.Fatalf("fixture needs data and coding shreds") } diff --git a/pkg/turbine/broadcast.go b/pkg/turbine/broadcast.go index fee229158..814951132 100644 --- a/pkg/turbine/broadcast.go +++ b/pkg/turbine/broadcast.go @@ -3,7 +3,6 @@ package turbine import ( "fmt" "net" - "sort" "sync" "github.com/gagliardetto/solana-go" @@ -116,8 +115,8 @@ type BroadcastSessionConfig struct { // It seeds the chained merkle root embedded in this slot's first FEC batch. ParentChainedMerkleRoot solana.Hash Broadcaster PacketBroadcaster - UserAgent []byte - Version uint16 + UserAgent []byte + Version uint16 } func NewBroadcastSession(cfg BroadcastSessionConfig) *BroadcastSession { @@ -192,7 +191,7 @@ func (s *BroadcastSession) broadcastComponent(component BlockComponent, isLastIn if s.broadcaster == nil { return nil } - batch, nextData, nextCode, err := s.shredder.MakeMerkleShredsFromComponent( + batch, nextData, nextCode, err := s.shredder.makeMerklePacketsFromComponent( s.leader, component, isLastInSlot, @@ -203,44 +202,9 @@ func (s *BroadcastSession) broadcastComponent(component BlockComponent, isLastIn if err != nil { return err } - s.chainedMerkleRoot = batch.ChainedMerkleRoot + s.chainedMerkleRoot = batch.chainedMerkleRoot s.nextDataIndex = nextData s.nextCodeIndex = nextCode - if len(batch.DataShreds) > 0 { - s.fecSetRoots = appendFECSetMerkleRoots(s.fecSetRoots, batch.DataShreds) - } - return s.broadcaster.Broadcast(batch.Packets) -} - -func appendFECSetMerkleRoots(roots []solana.Hash, dataShreds []*Shred) []solana.Hash { - if len(dataShreds) == 0 { - return roots - } - indices := make([]uint32, 0) - seen := make(map[uint32]struct{}) - for _, shred := range dataShreds { - if shred == nil { - continue - } - if _, ok := seen[shred.FECSetIndex]; ok { - continue - } - seen[shred.FECSetIndex] = struct{}{} - indices = append(indices, shred.FECSetIndex) - } - sort.Slice(indices, func(i, j int) bool { return indices[i] < indices[j] }) - for _, fecSetIndex := range indices { - for _, shred := range dataShreds { - if shred == nil || shred.FECSetIndex != fecSetIndex { - continue - } - root, err := shred.MerkleRoot() - if err != nil { - continue - } - roots = append(roots, root) - break - } - } - return roots + s.fecSetRoots = append(s.fecSetRoots, batch.fecSetRoots...) + return s.broadcaster.Broadcast(batch.packets) } diff --git a/pkg/turbine/broadcast_test.go b/pkg/turbine/broadcast_test.go index 09e82a397..11c211c76 100644 --- a/pkg/turbine/broadcast_test.go +++ b/pkg/turbine/broadcast_test.go @@ -52,6 +52,70 @@ func TestBroadcastSessionHeaderAndFooter(t *testing.T) { require.NotEqual(t, parentBlockID, chainedRoot) } +func TestBroadcastSessionMatchesParsedShreds(t *testing.T) { + leader := testBroadcastLeader(t) + parentID, parentRoot := solana.Hash{0xaa}, solana.Hash{0xbb} + capture := &packetCapture{} + session := NewBroadcastSession(BroadcastSessionConfig{ + Leader: leader, Slot: 100, ParentSlot: 99, Version: 7, + ParentChainedMerkleRoot: parentRoot, Broadcaster: capture, + }) + shredder := Shredder{Slot: 100, ParentSlot: 99, Version: 7} + txns := make([]solana.Transaction, 400) + for i := range txns { + txns[i] = mustParseTransferTx(t, uint64(i)) + } + entries, err := NewEntryBatch([]Entry{{NumHashes: 1, Hash: solana.Hash{1}, Txns: txns}}) + require.NoError(t, err) + tick, err := NewEntryBatch([]Entry{{NumHashes: 1, Hash: solana.Hash{2}}}) + require.NoError(t, err) + components := []BlockComponent{ + NewBlockHeader(99, parentID), + entries, // More than two FEC sets: every root must enter the block ID. + NewUpdateParent(98, solana.Hash{3}), + NewBlockFooter(BlockFooter{BankHash: solana.Hash{4}}), + tick, + } + var nextData, nextCode uint32 + root := parentRoot + var roots []solana.Hash + for i, component := range components { + last := i == len(components)-1 + batch, data, code, err := shredder.MakeMerkleShredsFromComponent( + leader, component, last, root, nextData, nextCode, + ) + require.NoError(t, err) + if i == 1 { + require.Greater(t, len(batch.DataShreds), 2*dataShredsPerFECBlock) + } + for j, shred := range batch.DataShreds { + if j == 0 || shred.FECSetIndex != batch.DataShreds[j-1].FECSetIndex { + fecRoot, err := shred.MerkleRoot() + require.NoError(t, err) + roots = append(roots, fecRoot) + } + } + capture.packets = nil + require.NoError(t, session.BroadcastComponent(component, last)) + require.Equal(t, batch.Packets, capture.packets) + require.Equal(t, batch.ChainedMerkleRoot, session.ChainedMerkleRoot()) + require.Equal(t, data, session.nextDataIndex) + require.Equal(t, code, session.nextCodeIndex) + require.Equal(t, roots, session.fecSetRoots) + require.Equal(t, DoubleMerkleBlockID(99, parentID, roots), session.BlockID(99, parentID)) + root, nextData, nextCode = batch.ChainedMerkleRoot, data, code + } + + // Invalid components must not publish packets or advance the commitment. + capture.packets = nil + require.Error(t, session.BroadcastComponent(BlockComponent{Marker: &BlockMarker{Kind: 255}}, false)) + require.Empty(t, capture.packets) + require.Equal(t, root, session.ChainedMerkleRoot()) + require.Equal(t, nextData, session.nextDataIndex) + require.Equal(t, nextCode, session.nextCodeIndex) + require.Equal(t, roots, session.fecSetRoots) +} + func TestUDPBroadcasterLoopback(t *testing.T) { recvAddr, err := net.ResolveUDPAddr("udp", "127.0.0.1:0") require.NoError(t, err) diff --git a/pkg/turbine/cancellation_regression_test.go b/pkg/turbine/cancellation_regression_test.go index 1f85e4df9..719931216 100644 --- a/pkg/turbine/cancellation_regression_test.go +++ b/pkg/turbine/cancellation_regression_test.go @@ -17,86 +17,52 @@ func TestTransactionVerifierCancellationDuringBlockedAdmissionJoinsAdmittedJobs( const workers = 2 blocker := verifierTestBlock(workers) target := verifierTestBlock(2 * workers) - filler := &solana.Transaction{} - release := make(chan struct{}) var releaseOnce sync.Once started := make(chan struct{}, workers) var targetFirstCalls atomic.Int32 var targetLaterCalls atomic.Int32 - - verifier := newTransactionVerifier(workers, workers, func(tx *solana.Transaction) error { + // Single-transaction groups make the blocked-admission boundary exact. + verifier := newTransactionVerifierWithBatchTarget(workers, 1, 1, func(tx *solana.Transaction) error { switch tx { case blocker.Transactions[0], blocker.Transactions[1]: started <- struct{}{} <-release case target.Transactions[0]: targetFirstCalls.Add(1) - case target.Transactions[1], target.Transactions[2], target.Transactions[3]: + default: targetLaterCalls.Add(1) } return nil }) - + defer verifier.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) ctx, cancel := context.WithCancel(context.Background()) + defer cancel() blockerDone := make(chan error, 1) targetDone := make(chan error, 1) - var calls sync.WaitGroup - calls.Add(1) - go func() { - defer calls.Done() - blockerDone <- verifier.verifyBlock(blocker) - }() + go func() { blockerDone <- verifier.verifyBlock(blocker) }() for range workers { - select { - case <-started: - case <-time.After(3 * time.Second): - cancel() - releaseOnce.Do(func() { close(release) }) - calls.Wait() - verifier.closeAndWait() - t.Fatal("timed out occupying transaction verifier workers") - } + waitSignal(t, started, "occupied verifier worker") } - // Leave one queued job ahead of the target. With both workers occupied and - // a two-entry queue, the target admits transaction 0 and then blocks trying - // to admit transaction 1. Cancellation must wait for transaction 0 to drain, - // while transactions 1 and all later chunks must never reach a worker. - var fillerErr error - var fillerDone sync.WaitGroup - fillerDone.Add(1) - verifier.jobs <- transactionVerifyJob{tx: filler, err: &fillerErr, done: &fillerDone} - - calls.Add(1) - go func() { - defer calls.Done() - targetDone <- verifier.verifyBlockContext(ctx, target) - }() + // Both workers are occupied. The target's first group fills the one-entry + // queue, and its second blocks on admission. Cancellation must still join + // the first group while preventing any later transactions from running. + go func() { targetDone <- verifier.verifyBlockContext(ctx, target) }() deadline := time.Now().Add(3 * time.Second) for len(verifier.jobs) != cap(verifier.jobs) { if time.Now().After(deadline) { - cancel() - releaseOnce.Do(func() { close(release) }) - calls.Wait() - fillerDone.Wait() - verifier.closeAndWait() - t.Fatalf("transaction queue did not fill: len=%d cap=%d", len(verifier.jobs), cap(verifier.jobs)) + t.Fatal("transaction queue did not fill") } time.Sleep(time.Millisecond) } - cancel() select { case err := <-targetDone: - releaseOnce.Do(func() { close(release) }) - calls.Wait() - fillerDone.Wait() - verifier.closeAndWait() t.Fatalf("canceled verifier returned before its admitted job joined: %v", err) case <-time.After(50 * time.Millisecond): } - releaseOnce.Do(func() { close(release) }) select { case err := <-targetDone: @@ -114,13 +80,6 @@ func TestTransactionVerifierCancellationDuringBlockedAdmissionJoinsAdmittedJobs( case <-time.After(3 * time.Second): t.Fatal("blocker verification did not drain") } - fillerDone.Wait() - calls.Wait() - verifier.closeAndWait() - - if fillerErr != nil { - t.Fatalf("filler verification: %v", fillerErr) - } if got := targetFirstCalls.Load(); got != 1 { t.Fatalf("admitted target transaction calls = %d, want 1", got) } diff --git a/pkg/turbine/child_repair.go b/pkg/turbine/child_repair.go new file mode 100644 index 000000000..786d3ee2d --- /dev/null +++ b/pkg/turbine/child_repair.go @@ -0,0 +1,110 @@ +package turbine + +import "time" + +const childRepairLimit = 4 +const childRepairLifetime = 2 * time.Second + +// SetStreamRepairParent anchors lookahead to the generation currently executing. +// Zero clears the hint on discard/finalize. This grants fetching, never execution +// or fork choice: the child's parent block ID may not be verifiable until full. +func (a *SlotAssembler) SetStreamRepairParent(g StreamGeneration) { + a.mu.Lock() + defer a.mu.Unlock() + a.streamRepairInvalidChild = nil + a.streamRepairParent = nil + a.streamRepairChild = nil + slot := g.Slot() + if g.IsZero() || slot == 0 || a.streamSubscriber == nil { + return + } + p := g.state + if a.streamStatusLocked(g) == StreamGone { + return + } + a.streamRepairParent = p + a.streamRepairUntil = time.Now().Add(childRepairLifetime) + // Reconcile a header published before replay installed the anchor. + if slot == ^uint64(0) { + return + } + c := a.slots[slot+1] + if c == nil || c.prefetch == nil { + return + } + b := c.prefetch.batches[0] + if b == nil || b.ready == nil { + return + } + select { + case <-b.ready: + a.noteChildRepairHeaderLocked(c, b) + default: + } +} + +// Called by the asynchronous decoder, not replay's busy execution goroutine. +func (a *SlotAssembler) noteChildRepairHeaderLocked(s *slotState, b *prefetchedShredBatch) { + p := a.streamRepairParent + if p == nil || s == nil || a.slots[s.slot] != s || b == nil || !b.marker || b.parent == nil { + return + } + if s == p && b.parent.FromUpdateParent { + a.streamRepairParent = nil + a.streamRepairChild = nil + return + } + if p.slot == ^uint64(0) || s.slot != p.slot+1 { + return + } + if b.parent.FromUpdateParent || b.parent.ParentSlot != p.slot { + a.streamRepairInvalidChild = s + a.streamRepairChild = nil + return + } + if a.streamRepairInvalidChild == s || b.start != 0 || b.err != nil || b.parent.ParentSlot != p.slot || a.streamRepairChild == s { + return + } + if time.Now().After(a.streamRepairUntil) || a.streamStatusLocked(StreamGeneration{slot: p.slot, state: p}) == StreamGone { + return + } + a.streamRepairChild = s + // Nonblocking channel send has no callback or lock acquisition. It uses the + // existing coalesced/minimum-spacing repair scheduler, including under a.mu. + select { + case a.streamRepairWake <- struct{}{}: + default: + } +} + +func (a *SlotAssembler) childRepairRequest(now time.Time) (SlotRepairRequest, bool) { + a.mu.Lock() + defer a.mu.Unlock() + p, c := a.streamRepairParent, a.streamRepairChild + if a.streamSubscriber == nil || p == nil || c == nil || now.After(a.streamRepairUntil) { + return SlotRepairRequest{}, false + } + if a.streamStatusLocked(StreamGeneration{slot: p.slot, state: p}) == StreamGone || a.slots[c.slot] != c || c.completing { + return SlotRepairRequest{}, false + } + req, ok := c.repairRequestWithPrefix(childRepairLimit, true) + if !ok || len(req.MissingDataShreds) == 0 { + return SlotRepairRequest{}, false + } + // Only the earliest span, not four unrelated holes or highest-index probes. + first := req.MissingDataShreds[0] + end := first + 1 + for _, f := range c.fecSets { + if f.haveLayout && f.fecSetIndex <= first && first < f.fecSetIndex+uint32(f.layout.dataShreds) { + end = f.fecSetIndex + uint32(f.layout.dataShreds) + break + } + } + n := 1 + for n < len(req.MissingDataShreds) && req.MissingDataShreds[n] < end { + n++ + } + req.MissingDataShreds = req.MissingDataShreds[:n] + req.NeedHighestDataShred = false + return req, true +} diff --git a/pkg/turbine/child_repair_test.go b/pkg/turbine/child_repair_test.go new file mode 100644 index 000000000..b6718bd97 --- /dev/null +++ b/pkg/turbine/child_repair_test.go @@ -0,0 +1,160 @@ +package turbine + +import ( + "errors" + "github.com/Overclock-Validator/mithril/pkg/gossip" + "github.com/stretchr/testify/require" + "net" + "testing" + "time" +) + +func childRepairFixture(t *testing.T) (*SlotAssembler, *slotState, *slotState, *prefetchedShredBatch) { + t.Helper() + a := NewSlotAssembler() + a.SubscribeStream(make(chan StreamEvent, 1)) + p := newRepairSelectionSlot(100) + c := newRepairSelectionSlot(101) + addCodedSet(c, 0, 32, 32, seq(0, 19), 8) // deficit4 + addCodedSet(c, 32, 32, 32, seq(32, 56), 6) // cheaper later span + a.slots[100] = p + a.slots[101] = c + a.maxObservedSlot = 101 + done := make(chan struct{}) + close(done) + b := &prefetchedShredBatch{start: 0, marker: true, parent: &AlpenglowParentInfo{ParentSlot: 100}, ready: done} + c.prefetch = &slotEntryPrefetch{batches: map[uint32]*prefetchedShredBatch{0: b}} + a.SetStreamRepairParent(StreamGeneration{slot: 100, state: p}) + return a, p, c, b +} +func TestChildRepairEarlyHeaderAndDroppedEvents(t *testing.T) { + a, _, c, b := childRepairFixture(t) + req, ok := a.childRepairRequest(time.Now()) + require.True(t, ok) + require.Equal(t, []uint32{20, 21, 22, 23}, req.MissingDataShreds) + require.False(t, req.NeedHighestDataShred) + a.streamRepairChild = nil + a.streamSubscriber <- StreamEvent{} // full notification channel + wake := make(chan struct{}, 1) + a.streamRepairWake = wake + a.mu.Lock() + a.publishStreamBatchReadyLocked(c, b) + a.mu.Unlock() + require.Len(t, wake, 1) + require.Equal(t, uint64(1), a.StreamDroppedEvents()) + _, ok = a.childRepairRequest(time.Now()) + require.True(t, ok) + // Repeated publication is not a new repair wakeup. + <-wake + a.mu.Lock() + a.publishStreamBatchReadyLocked(c, b) + a.mu.Unlock() + require.Empty(t, wake) +} +func TestChildRepairLifecycle(t *testing.T) { + for _, kind := range []string{"expiry", "clear", "child-reset", "parent-reset", "child-complete", "parent-replaced", "unsubscribe", "update-parent", "wrong-parent"} { + t.Run(kind, func(t *testing.T) { + a, p, c, b := childRepairFixture(t) + switch kind { + case "expiry": + a.streamRepairUntil = time.Now().Add(-time.Second) + case "clear": + a.SetStreamRepairParent(StreamGeneration{}) + case "child-reset": + c.prefetch = nil // fixture has no live prefetch workers + a.ResetSlot(c.slot) + case "parent-reset": + a.ResetSlot(p.slot) + case "child-complete": + c.completing = true + case "parent-replaced": + a.slots[p.slot] = newRepairSelectionSlot(p.slot) + case "unsubscribe": + a.SubscribeStream(nil) + case "update-parent", "wrong-parent": + changed := prefetchedShredBatch{start: b.start, end: b.end, parent: b.parent, marker: b.marker, ready: b.ready} + parent := *b.parent + changed.parent = &parent + if kind == "update-parent" { + changed.start = 32 + parent.FromUpdateParent = true + } else { + parent.ParentSlot = 99 + } + a.mu.Lock() + a.noteChildRepairHeaderLocked(c, &changed) + a.noteChildRepairHeaderLocked(c, b) + a.mu.Unlock() + } + _, ok := a.childRepairRequest(time.Now()) + require.False(t, ok) + }) + } + // Parent assembly can finish while replay is still executing it. + a, p, _, _ := childRepairFixture(t) + delete(a.slots, p.slot) + p.streamCompleted = true + _, ok := a.childRepairRequest(time.Now()) + require.True(t, ok) + a.ResetSlot(p.slot) + _, ok = a.childRepairRequest(time.Now()) + require.False(t, ok) +} +func TestChildRepairRejectsUnrelatedAndInvalidHeader(t *testing.T) { + a, _, c, b := childRepairFixture(t) + a.streamRepairChild = nil + bad := prefetchedShredBatch{start: b.start, end: b.end, parent: b.parent, marker: b.marker, ready: b.ready} + bad.err = errors.New("invalid header") + a.mu.Lock() + a.noteChildRepairHeaderLocked(c, &bad) + a.mu.Unlock() + _, ok := a.childRepairRequest(time.Now()) + require.False(t, ok) + other := newRepairSelectionSlot(102) + a.mu.Lock() + a.noteChildRepairHeaderLocked(other, b) + a.mu.Unlock() + _, ok = a.childRepairRequest(time.Now()) + require.False(t, ok) +} +func TestChildRepairUsesOnlyRemainingBudget(t *testing.T) { + sink, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer sink.Close() + conn, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer conn.Close() + c := newPacingTestClient(t) + peers := []gossip.RepairPeer{{Addr: sink.LocalAddr().(*net.UDPAddr)}} + req := SlotRepairRequest{Slot: 101, MissingDataShreds: []uint32{20, 21, 22, 23, 24, 25}, NeedHighestDataShred: true} + require.Zero(t, c.sendChildRepair(conn, peers, req, 0, time.Second)) + require.Equal(t, 2, c.sendChildRepair(conn, peers, req, 2, time.Second)) + require.Equal(t, 2, c.sendChildRepair(conn, peers, req, 100, time.Second)) + require.Zero(t, c.sendChildRepair(conn, peers, req, 100, time.Second)) + require.Len(t, c.outstanding, 4) + for k := range c.outstanding { + require.Equal(t, repairRequestWindowIndex, k.kind) + require.Zero(t, k.attempt) + } +} + +func TestChildRepairRejectsStaleParentAnchor(t *testing.T) { + a, p, _, _ := childRepairFixture(t) + a.slots[p.slot] = newRepairSelectionSlot(p.slot) + a.SetStreamRepairParent(StreamGeneration{slot: p.slot, state: p}) + require.Nil(t, a.streamRepairParent) +} + +func TestChildRepairParentUpdateClearsLookahead(t *testing.T) { + a, p, _, b := childRepairFixture(t) + update := prefetchedShredBatch{start: b.start, end: b.end, parent: b.parent, marker: b.marker, ready: b.ready} + info := *b.parent + info.FromUpdateParent = true + update.parent = &info + update.start = 32 + a.mu.Lock() + a.noteChildRepairHeaderLocked(p, &update) + a.mu.Unlock() + _, ok := a.childRepairRequest(time.Now()) + require.False(t, ok) +} diff --git a/pkg/turbine/cluster_nodes.go b/pkg/turbine/cluster_nodes.go index 8446a3966..ff6e0bdc0 100644 --- a/pkg/turbine/cluster_nodes.go +++ b/pkg/turbine/cluster_nodes.go @@ -89,6 +89,13 @@ func newClusterNodes(cfg ClusterNodesConfig, broadcast bool) *ClusterNodes { // shuffle because it already owns the shred. Tree placement and fanout match // Agave ClusterNodes::get_retransmit_addrs. func (c *ClusterNodes) RetransmitPeers(leader solana.PublicKey, shred ShredID, fanout int) (uint8, []*net.UDPAddr, error) { + return c.retransmitPeersInto(leader, shred, fanout, nil) +} + +// retransmitPeersInto uses caller-owned result storage. The caller must finish +// using the returned slice before reusing dst; addresses still belong to this +// immutable cluster snapshot. Public callers retain the allocating API above. +func (c *ClusterNodes) retransmitPeersInto(leader solana.PublicKey, shred ShredID, fanout int, dst []*net.UDPAddr) (uint8, []*net.UDPAddr, error) { if c == nil || fanout <= 0 { return maxTurbineHops - 1, nil, nil } @@ -123,7 +130,10 @@ func (c *ClusterNodes) RetransmitPeers(leader solana.PublicKey, shred ShredID, f step = 1 } position := anchor*fanout + offset + 1 - peers := make([]*net.UDPAddr, 0, fanout) + peers := dst[:0] + if dst == nil { + peers = make([]*net.UDPAddr, 0, fanout) + } shufflePosition := selfPos for range fanout { var index int diff --git a/pkg/turbine/completion_order_root_test.go b/pkg/turbine/completion_order_root_test.go new file mode 100644 index 000000000..5fd8fd7f7 --- /dev/null +++ b/pkg/turbine/completion_order_root_test.go @@ -0,0 +1,228 @@ +package turbine + +import ( + "bytes" + "context" + "errors" + "sort" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestCompletedShredOrderPreservesSparseAndExtraTails(t *testing.T) { + for _, indexes := range [][]uint32{{0, 1, 2}, {0, 2, 3}, {0, 1, 2, 9}, {1, 3, 7}} { + s := &slotState{haveLast: true, lastIndex: 2, shreds: make(map[uint32]*Shred)} + for _, idx := range indexes { + s.shreds[idx] = &Shred{Index: idx, Type: ShredTypeData} + } + got := s.orderedShreds() + require.Len(t, got, len(indexes)) + for i, idx := range indexes { + require.Same(t, s.shreds[idx], got[i]) + } + } +} + +func TestOrderedEntryDecodeMatchesPublicUnorderedDecode(t *testing.T) { + packets := agavePaddedSlot1752420Packets(t) + s := &slotState{shreds: make(map[uint32]*Shred)} + var shuffled []*Shred + for _, packet := range packets { + shred, err := ParseShred(packet) + require.NoError(t, err) + s.shreds[shred.Index] = shred + shuffled = append(shuffled, shred) + } + for i, j := 0, len(shuffled)-1; i < j; i, j = i+1, j-1 { + shuffled[i], shuffled[j] = shuffled[j], shuffled[i] + } + want, parent, footer, err := DecodeEntriesAndAlpenglowMarkersFromDataShreds(shuffled) + require.NoError(t, err) + got, gotParent, gotFooter, err := decodeEntriesFromOrderedDataShreds(s.orderedShreds(), nil) + require.NoError(t, err) + require.Equal(t, want, got) + require.Equal(t, parent, gotParent) + require.Equal(t, footer, gotFooter) +} + +func TestAuthenticatedFECRootRejectsChangedInputAndReplacement(t *testing.T) { + shred, leader := buildSignedTestShred(t, 100, 42) + var verifier shredSigCache + root, err := verifier.verifyShredRoot(shred, leader) + require.NoError(t, err) + f := &fecState{data: map[uint32]*Shred{0: shred}} + f.rememberAuthenticatedRoot(shred, root) + cached := f.rootCache + require.True(t, cached.matches(shred)) + for i := range shred.Payload { + shred.Payload[i] ^= 1 + require.False(t, cached.matches(shred), "payload byte %d", i) + shred.Payload[i] ^= 1 + } + for _, mutate := range []func(*Shred){ + func(s *Shred) { s.Variant ^= 1 }, func(s *Shred) { s.Type = ShredTypeCode }, + func(s *Shred) { s.Index++ }, func(s *Shred) { s.FECSetIndex++ }, + func(s *Shred) { s.NumDataShreds++ }, func(s *Shred) { s.Position++ }, + } { + original := *shred + mutate(shred) + require.False(t, cached.matches(shred)) + *shred = original + } + shred.Payload[dataHeaderSize] ^= 1 + want, err := shred.MerkleRoot() + require.NoError(t, err) + got, ok, err := f.merkleRoot() + require.NoError(t, err) + require.True(t, ok) + require.Equal(t, want, got) + require.NotEqual(t, root, got) + _, err = verifier.verifyShredRoot(shred, leader) + require.ErrorIs(t, err, ErrInvalidSignature) + shred.Payload[dataHeaderSize] ^= 1 + replacement := *shred + require.False(t, cached.matches(&replacement)) + f.data[0] = &replacement + got, ok, err = f.merkleRoot() + require.NoError(t, err) + require.True(t, ok) + require.Equal(t, root, got) +} + +func TestAuthenticatedFECRootAdmissionAndReset(t *testing.T) { + shred, leader := buildSignedTestShred(t, 100, 42) + var verifier shredSigCache + root, err := verifier.verifyShredRoot(shred, leader) + require.NoError(t, err) + a := NewSlotAssembler() + work, err := a.addShredFromWithRoot(shred, false, &root) + require.NoError(t, err) + require.NotNil(t, work) + f := work.state.fecSets[0] + require.True(t, f.rootCache.matches(shred)) + // A duplicate cannot overwrite the admitted root, even while completion is + // retried after cancellation. Reset starts a separate FEC generation. + a.abortCompletion(work) + wrong := solana.Hash{99} + _, err = a.addShredFromWithRoot(shred, true, &wrong) + require.NoError(t, err) + require.Equal(t, root, f.rootCache.root) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + require.True(t, a.processCompletion(ctx, work).canceled) + a.ResetSlot(100) + work, err = a.addShredFrom(shred, false) + require.NoError(t, err) + require.NotNil(t, work) + require.Nil(t, work.state.fecSets[0].rootCache) +} + +func referenceFECRoot(f *fecState) (solana.Hash, bool, error) { + for _, idx := range sortedUint32Keys(f.data) { + s := f.data[idx] + if s == nil || s.Recovered { + continue + } + root, err := s.MerkleRoot() + if errors.Is(err, ErrUnsupportedShred) { + continue + } + return root, err == nil, err + } + for _, idx := range sortedUint16Keys(f.coding) { + s := f.coding[idx] + if s == nil { + continue + } + root, err := s.MerkleRoot() + if errors.Is(err, ErrUnsupportedShred) { + continue + } + return root, err == nil, err + } + return solana.Hash{}, false, nil +} + +func FuzzFECRootSelectionMatchesSortedReference(f *testing.F) { + f.Add([]byte{0, 1, 2, 3, 4, 5, 6, 7}) + f.Add([]byte{9, 0, 0, 0, 2, 1, 3, 7}) + f.Fuzz(func(t *testing.T, choices []byte) { + if len(choices) > 512 { + t.Skip() + } + state := &fecState{data: make(map[uint32]*Shred), coding: make(map[uint16]*Shred)} + for i, choice := range choices { + s := &Shred{Variant: merkleDataVariant, Type: ShredType(choice % 3), Payload: bytes.Repeat([]byte{choice}, dataPayloadSize)} + switch choice % 5 { + case 0: + s = nil + case 1: + s.Recovered = true + case 2: + s.Variant = legacyDataVariant + case 3: + s.Payload = s.Payload[:20] + } + if i%2 == 0 { + state.data[uint32(choice)] = s + } else { + state.coding[uint16(choice)] = s + } + } + want, wantOK, wantErr := referenceFECRoot(state) + got, gotOK, gotErr := state.merkleRoot() + require.Equal(t, want, got) + require.Equal(t, wantOK, gotOK) + if wantErr == nil { + require.NoError(t, gotErr) + } else { + require.EqualError(t, gotErr, wantErr.Error()) + } + }) +} + +func TestFECRootCacheKeepsDeterministicDataPrecedence(t *testing.T) { + packets := append(localnetMerkleShreds(t, "d"), localnetMerkleShreds(t, "c")...) + f := &fecState{data: make(map[uint32]*Shred), coding: make(map[uint16]*Shred)} + var shreds []*Shred + for _, packet := range packets { + s, err := ParseShred(packet) + require.NoError(t, err) + if s.FECSetIndex != 0 { + continue + } + shreds = append(shreds, s) + } + require.NotEmpty(t, shreds) + sort.Slice(shreds, func(i, j int) bool { + if shreds[i].Type != shreds[j].Type { + return shreds[i].Type == ShredTypeCode + } + return shreds[i].Index > shreds[j].Index + }) + for _, s := range shreds { + root, err := s.MerkleRoot() + require.NoError(t, err) + if s.Type == ShredTypeData { + f.data[s.Index-s.FECSetIndex] = s + } else { + f.coding[s.Position] = s + } + f.rememberAuthenticatedRoot(s, root) + want, wantOK, wantErr := referenceFECRoot(f) + got, gotOK, gotErr := f.merkleRoot() + require.Equal(t, wantErr, gotErr) + require.Equal(t, wantOK, gotOK) + require.Equal(t, want, got) + } + for _, s := range f.data { + s.Recovered = true + } + want, wantOK, wantErr := referenceFECRoot(f) + got, gotOK, gotErr := f.merkleRoot() + require.Equal(t, wantErr, gotErr) + require.Equal(t, wantOK, gotOK) + require.Equal(t, want, got) +} diff --git a/pkg/turbine/completion_pool.go b/pkg/turbine/completion_pool.go index 451b0095c..e493b681d 100644 --- a/pkg/turbine/completion_pool.go +++ b/pkg/turbine/completion_pool.go @@ -83,10 +83,11 @@ func newSlotCompletionPool(assembler *SlotAssembler, resetGate *sync.RWMutex, on p.resetGate.RUnlock() } p.results <- slotCompletionResult{ - block: blk, - err: err, - hydrated: queued.hydrated, - pending: pending, + generation: StreamGeneration{slot: queued.work.state.slot, state: queued.work.state}, + block: blk, + err: err, + hydrated: queued.hydrated, + pending: pending, } } }() diff --git a/pkg/turbine/component_shredder.go b/pkg/turbine/component_shredder.go index ad4c9229b..109408a4c 100644 --- a/pkg/turbine/component_shredder.go +++ b/pkg/turbine/component_shredder.go @@ -14,7 +14,7 @@ type Shredder struct { ReferenceTick uint8 } -// ShredBatch is one FEC batch emitted for a single block component. +// ShredBatch contains the FEC batches emitted for a single block component. type ShredBatch struct { Slot uint64 Component BlockComponent @@ -34,23 +34,8 @@ func (s *Shredder) MakeMerkleShredsFromComponent( nextShredIndex uint32, nextCodeIndex uint32, ) (ShredBatch, uint32, uint32, error) { - bytes, err := MarshalBlockComponent(component) - if err != nil { - return ShredBatch{}, nextShredIndex, nextCodeIndex, err - } - gen := ShredGenerator{ - Slot: s.Slot, - ParentSlot: s.ParentSlot, - Version: s.Version, - ReferenceTick: s.ReferenceTick, - } - packets, root, nextData, nextCode, err := gen.MakeShredsFromData( - leader, - bytes, - isLastInSlot, - chainedMerkleRoot, - nextShredIndex, - nextCodeIndex, + generated, nextData, nextCode, err := s.makeMerklePacketsFromComponent( + leader, component, isLastInSlot, chainedMerkleRoot, nextShredIndex, nextCodeIndex, ) if err != nil { return ShredBatch{}, nextShredIndex, nextCodeIndex, err @@ -58,11 +43,11 @@ func (s *Shredder) MakeMerkleShredsFromComponent( batch := ShredBatch{ Slot: s.Slot, Component: component, - Packets: packets, - ChainedMerkleRoot: root, + Packets: generated.packets, + ChainedMerkleRoot: generated.chainedMerkleRoot, IsLastInSlot: isLastInSlot, } - for _, packet := range packets { + for _, packet := range batch.Packets { shred, err := ParseShred(packet) if err != nil { return ShredBatch{}, nextShredIndex, nextCodeIndex, fmt.Errorf("parse generated shred: %w", err) @@ -75,3 +60,37 @@ func (s *Shredder) MakeMerkleShredsFromComponent( } return batch, nextData, nextCode, nil } + +// makeMerklePacketsFromComponent serves the producer, which needs the wire +// packets and FEC roots but not the owning Shred objects exposed by the public API. +func (s *Shredder) makeMerklePacketsFromComponent( + leader solana.PrivateKey, + component BlockComponent, + isLastInSlot bool, + chainedMerkleRoot solana.Hash, + nextShredIndex uint32, + nextCodeIndex uint32, +) (shredPackets, uint32, uint32, error) { + bytes, err := MarshalBlockComponent(component) + if err != nil { + return shredPackets{}, nextShredIndex, nextCodeIndex, err + } + gen := ShredGenerator{ + Slot: s.Slot, + ParentSlot: s.ParentSlot, + Version: s.Version, + ReferenceTick: s.ReferenceTick, + } + batch, nextData, nextCode, err := gen.makeShredsFromData( + leader, + bytes, + isLastInSlot, + chainedMerkleRoot, + nextShredIndex, + nextCodeIndex, + ) + if err != nil { + return shredPackets{}, nextShredIndex, nextCodeIndex, err + } + return batch, nextData, nextCode, nil +} diff --git a/pkg/turbine/component_test.go b/pkg/turbine/component_test.go index 64b05760c..2617715be 100644 --- a/pkg/turbine/component_test.go +++ b/pkg/turbine/component_test.go @@ -131,6 +131,36 @@ func TestShredEntryBatchRoundTrip(t *testing.T) { require.Equal(t, entry.NumHashes, components[0].EntryBatch[0].NumHashes) } +func TestShredMultiFECEntryBatchRoundTrip(t *testing.T) { + leader := testLeader(t) + entries := make([]turbine.Entry, 1300) + for i := range entries { + entries[i] = turbine.Entry{NumHashes: 1, Hash: solana.Hash{byte(i), byte(i >> 8)}} + } + component, err := turbine.NewEntryBatch(entries) + require.NoError(t, err) + + shredder := turbine.Shredder{Slot: 100, ParentSlot: 99, Version: 42, ReferenceTick: 63} + batch, _, _, err := shredder.MakeMerkleShredsFromComponent( + leader, component, true, solana.Hash{}, 0, 0, + ) + require.NoError(t, err) + require.Greater(t, len(batch.DataShreds), 32) + for i, shred := range batch.DataShreds[:len(batch.DataShreds)-1] { + require.False(t, shred.DataComplete(), "intermediate data shred %d ended the component", i) + } + require.True(t, batch.DataShreds[len(batch.DataShreds)-1].DataComplete()) + require.True(t, batch.DataShreds[len(batch.DataShreds)-1].LastInSlot()) + + components, err := turbine.DecodeComponentsFromDataShreds(batch.DataShreds) + require.NoError(t, err) + require.Len(t, components, 1) + require.Len(t, components[0].EntryBatch, len(entries)) + for i := range entries { + require.Equal(t, entries[i].Hash, components[0].EntryBatch[i].Hash) + } +} + func TestShredBlockHeaderMarkerRoundTrip(t *testing.T) { leader := testLeader(t) parentID := solana.Hash{8} diff --git a/pkg/turbine/entries.go b/pkg/turbine/entries.go index 9f6251318..e9372204d 100644 --- a/pkg/turbine/entries.go +++ b/pkg/turbine/entries.go @@ -1,6 +1,8 @@ package turbine import ( + "bytes" + "context" "encoding/binary" "fmt" "sort" @@ -40,6 +42,12 @@ type AlpenglowParentInfo struct { type entryDecodeTimings struct { transactionParse time.Duration + ctx context.Context + prefetched map[uint32]*prefetchedShredBatch + retained []*prefetchedShredBatch + all []*prefetchedShredBatch + prefetchWait time.Duration + traceFallback *transactionVerification } func (e *Entry) UnmarshalWithDecoder(decoder *bin.Decoder) error { @@ -124,69 +132,109 @@ func decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds []*Shred, timings *en sort.Slice(shreds, func(i, j int) bool { return shreds[i].Index < shreds[j].Index }) + return decodeEntriesFromOrderedDataShreds(shreds, timings) +} - type decodedEntryBatch struct { - start uint32 - entries []Entry - } - var entryBatches []decodedEntryBatch +// decodeEntriesFromOrderedDataShreds requires increasing shred indexes. The +// assembler supplies that order directly; public decoding still sorts input. +func decodeEntriesFromOrderedDataShreds(shreds []*Shred, timings *entryDecodeTimings) ([]Entry, *AlpenglowParentInfo, *BlockFooter, error) { + var entryBatches []*prefetchedShredBatch var parentInfo *AlpenglowParentInfo var blockFooter *BlockFooter - var batchBytes []byte var batchStart uint32 + var batchStartPos, batchSize int var haveBatch bool - for _, shred := range shreds { + for shredPos, shred := range shreds { if shred == nil || shred.Type != ShredTypeData { continue } if !haveBatch { batchStart = shred.Index + batchStartPos = shredPos haveBatch = true } - batchBytes = append(batchBytes, shred.Data...) + batchSize += len(shred.Data) if !shred.DataComplete() { continue } - if parent, footer, ok, err := decodeAlpenglowMarkerFromShredBatch(batchBytes, batchStart); err != nil { - return nil, nil, nil, fmt.Errorf("decode alpenglow block marker ending at shred %d: %w", shred.Index, err) - } else if ok { - if parent != nil { - parentInfo, err = mergeAlpenglowParentInfo(parentInfo, parent) - if err != nil { - return nil, nil, nil, fmt.Errorf("merge alpenglow parent marker ending at shred %d: %w", shred.Index, err) + batchShreds := shreds[batchStartPos : shredPos+1] + var batch *prefetchedShredBatch + if timings != nil { + if cached := timings.prefetched[batchStart]; cached != nil && cached.start == batchStart && cached.end == shred.Index { + ctx := timings.ctx + if ctx == nil { + ctx = context.Background() + } + if cached.ready != nil { + waitStarted := time.Now() + select { + case <-cached.ready: + case <-ctx.Done(): + timings.prefetchWait += time.Since(waitStarted) + return nil, nil, nil, ctx.Err() + } + timings.prefetchWait += time.Since(waitStarted) + } + if err := ctx.Err(); err != nil { + return nil, nil, nil, err + } + // Bounds alone cannot prove identity after repair or replacement. + // Read decoded fields only after the preparation channel closes. + // Compare the original slices directly: a cache hit needs no + // second component buffer or copies of already decoded bytes. + if len(cached.raw) == batchSize && dataShredBatchMatches(batchShreds, cached.raw) { + batch = cached } } - if footer != nil { - blockFooter = footer + } + if batch == nil { + // A miss owns a fresh, exactly sized backing array. Transactions + // retain instruction-data slices into it after this call returns. + var traceStart int64 + if timings != nil && entryTraceContext(timings.ctx) { + traceStart = entryTraceNow() + } + batchBytes := make([]byte, 0, batchSize) + for _, part := range batchShreds { + if part != nil && part.Type == ShredTypeData { + batchBytes = append(batchBytes, part.Data...) + } + } + batch = decodeClosedShredBatch(batchBytes, batchStart, shred.Index) + if traceStart != 0 { + batch.traceDecodeStart, batch.traceDecodeEnd = traceStart, entryTraceNow() + } + if timings != nil { + timings.transactionParse += batch.parseDuration } - batchBytes = nil - haveBatch = false - continue } - parseStart := time.Now() - batchEntries, consumed, err := decodeEntryBatchPrefix(batchBytes) if timings != nil { - timings.transactionParse += time.Since(parseStart) + timings.all = append(timings.all, batch) } - // A zero entry count with more bytes denotes a marker. Preserve the - // rejection of unrecognized or misplaced markers in this fallback path. - if err == nil && len(batchEntries) == 0 && consumed != len(batchBytes) { - err = fmt.Errorf("entry batch has %d trailing bytes", len(batchBytes)-consumed) + if batch.err != nil { + return nil, nil, nil, batch.err } - if err != nil { - return nil, nil, nil, fmt.Errorf("decode entry batch ending at shred %d: %w", shred.Index, err) + if batch.marker { + if batch.parent != nil { + var err error + parentInfo, err = mergeAlpenglowParentInfo(parentInfo, batch.parent) + if err != nil { + return nil, nil, nil, fmt.Errorf("merge alpenglow parent marker ending at shred %d: %w", shred.Index, err) + } + } + if batch.footer != nil { + blockFooter = batch.footer + } + batchSize = 0 + haveBatch = false + continue } - entryBatches = append(entryBatches, decodedEntryBatch{ - start: batchStart, - entries: batchEntries, - }) - // Decoded transactions retain slices into the batch buffer for instruction data. - // Keep the backing array alive instead of reusing and overwriting it. - batchBytes = nil + entryBatches = append(entryBatches, batch) + batchSize = 0 haveBatch = false } - if len(batchBytes) != 0 { - return nil, nil, nil, fmt.Errorf("slot ended with %d undecoded entry bytes", len(batchBytes)) + if batchSize != 0 { + return nil, nil, nil, fmt.Errorf("slot ended with %d undecoded entry bytes", batchSize) } var entries []Entry for _, batch := range entryBatches { @@ -199,10 +247,60 @@ func decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds []*Shred, timings *en continue } entries = append(entries, batch.entries...) + if timings != nil { + timings.retained = append(timings.retained, batch) + } } return entries, parentInfo, blockFooter, nil } +// dataShredBatchMatches is equivalent to comparing raw with the concatenated +// data bytes, including all padding. Neither equal prefixes nor changed lengths +// can reuse a cached signature verdict. It does not retain or allocate buffers. +func dataShredBatchMatches(shreds []*Shred, raw []byte) bool { + offset := 0 + for _, shred := range shreds { + if shred == nil || shred.Type != ShredTypeData { + continue + } + if len(shred.Data) > len(raw)-offset || !bytes.Equal(shred.Data, raw[offset:offset+len(shred.Data)]) { + return false + } + offset += len(shred.Data) + } + return offset == len(raw) +} + +// decodeClosedShredBatch owns raw through the returned decoded transactions. +// It decodes the same padded component envelope for early and full assembly; +// marker merging and UpdateParent selection still require the full slot. +func decodeClosedShredBatch(raw []byte, start, end uint32) *prefetchedShredBatch { + batch := &prefetchedShredBatch{start: start, end: end, raw: raw} + parent, footer, marker, err := decodeAlpenglowMarkerFromShredBatch(raw, start) + if err != nil { + batch.err = fmt.Errorf("decode alpenglow block marker ending at shred %d: %w", end, err) + return batch + } + if marker { + batch.parent, batch.footer, batch.marker = parent, footer, true + return batch + } + parseStarted := time.Now() + entries, consumed, err := decodeEntryBatchPrefix(raw) + batch.parseDuration = time.Since(parseStarted) + // A zero entry count with more bytes denotes a marker. Preserve rejection + // of unknown or misplaced markers instead of treating them as an empty batch. + if err == nil && len(entries) == 0 && consumed != len(raw) { + err = fmt.Errorf("entry batch has %d trailing bytes", len(raw)-consumed) + } + if err != nil { + batch.err = fmt.Errorf("decode entry batch ending at shred %d: %w", end, err) + return batch + } + batch.entries = entries + return batch +} + func decodeEntryBatch(data []byte) ([]Entry, error) { entries, consumed, err := decodeEntryBatchPrefix(data) if err != nil { diff --git a/pkg/turbine/entries_direct_compare_test.go b/pkg/turbine/entries_direct_compare_test.go new file mode 100644 index 000000000..4fb55637e --- /dev/null +++ b/pkg/turbine/entries_direct_compare_test.go @@ -0,0 +1,100 @@ +package turbine + +import ( + "bytes" + "context" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestDataShredBatchMatchesChecksLengthsAndEverySlice(t *testing.T) { + raw := []byte("0123456789abcdefghijklmnopqrstuvwxyz") + shreds := []*Shred{ + {Type: ShredTypeData, Data: raw[:10]}, + nil, + {Type: ShredTypeCode, Data: []byte("coding bytes are not component data")}, + {Type: ShredTypeData}, + {Type: ShredTypeData, Data: raw[10:20]}, + {Type: ShredTypeData, Data: raw[20:]}, + } + require.True(t, dataShredBatchMatches(shreds, bytes.Clone(raw))) + for i := range raw { + changed := bytes.Clone(raw) + changed[i] ^= 1 + require.False(t, dataShredBatchMatches(shreds, changed), "changed byte %d", i) + } + for i := 0; i < len(raw); i++ { + require.False(t, dataShredBatchMatches(shreds, raw[:i]), "truncated at %d", i) + } + require.False(t, dataShredBatchMatches(shreds, append(bytes.Clone(raw), 0))) + require.True(t, dataShredBatchMatches(nil, nil)) + require.False(t, dataShredBatchMatches(nil, raw)) +} + +func TestEntryDecodeDirectComparisonCannotReuseChangedSignatureVerdict(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + txs := verifierSignedTransactions(t, 1) + raw := prefetchTestPayload(t, txs) + cache := decodedPrefetchForTest(paddedComponentShreds(t, raw, 0, 0xa5)) + cached := cache[0] + var err error + cached.verification, err = v.submitTransactions(context.Background(), entryBatchTransactions(cached.entries)) + require.NoError(t, err) + _, err = cached.verification.wait() + require.NoError(t, err) + + bad := *txs[0] + bad.Signatures = append([]solana.Signature(nil), bad.Signatures...) + bad.Signatures[0][0] ^= 1 + changed := prefetchTestPayload(t, []*solana.Transaction{&bad}) + require.Len(t, changed, len(raw)) + timings := entryDecodeTimings{prefetched: cache} + entries, _, _, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(paddedComponentShreds(t, changed, 0, 0xa5), &timings) + require.NoError(t, err) + require.Len(t, timings.retained, 1) + require.NotSame(t, cached, timings.retained[0]) + require.Nil(t, timings.retained[0].verification) + blk := BlockFromEntries(100, 99, entries) + require.Equal(t, bad.Signatures[0], blk.Transactions[0].Signatures[0]) + require.ErrorContains(t, verifyDecodedEntryBatches(context.Background(), blk, timings.retained, v), "failed signature verification") + require.False(t, blk.TransactionSignaturesVerified()) +} + +func TestEntryDecodeDirectComparisonRejectsChangedCachedLength(t *testing.T) { + raw := prefetchTestPayload(t, verifierSignedTransactions(t, 1)) + for _, delta := range []int{-1, 1} { + shreds := paddedComponentShreds(t, raw, 0, 0xa5) + cache := decodedPrefetchForTest(shreds) + cached := cache[0] + if delta < 0 { + cached.raw = cached.raw[:len(cached.raw)-1] + } else { + cached.raw = append(cached.raw, 0) + } + timings := entryDecodeTimings{prefetched: cache} + entries, _, _, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.NoError(t, err) + require.Len(t, entries, 1) + require.NotSame(t, cached, timings.retained[0]) + } +} + +func FuzzDataShredBatchMatchesConcatenatedBytes(f *testing.F) { + f.Add([]byte("a component spanning several shreds"), []byte("a component spanning several shreds"), uint8(3)) + f.Add([]byte("truncated"), []byte("truncate"), uint8(1)) + f.Add([]byte{}, []byte{}, uint8(0)) + f.Fuzz(func(t *testing.T, data, candidate []byte, split uint8) { + if len(data) > 64*1024 || len(candidate) > 64*1024 { + t.Skip() + } + shreds := []*Shred{nil, {Type: ShredTypeCode, Data: []byte{1, 2, 3}}} + width := int(split) + 1 + for start := 0; start < len(data); start += width { + shreds = append(shreds, &Shred{Type: ShredTypeData, Data: data[start:min(start+width, len(data))]}) + } + require.Equal(t, bytes.Equal(data, candidate), dataShredBatchMatches(shreds, candidate)) + }) +} diff --git a/pkg/turbine/entries_prefetch_test.go b/pkg/turbine/entries_prefetch_test.go new file mode 100644 index 000000000..9ecdbf599 --- /dev/null +++ b/pkg/turbine/entries_prefetch_test.go @@ -0,0 +1,146 @@ +package turbine + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func decodedPrefetchForTest(shreds []*Shred) map[uint32]*prefetchedShredBatch { + cache := make(map[uint32]*prefetchedShredBatch) + var raw []byte + var start uint32 + for _, shred := range shreds { + if raw == nil { + start = shred.Index + } + raw = append(raw, shred.Data...) + if shred.DataComplete() { + batch := decodeClosedShredBatch(raw, start, shred.Index) + batch.ready = make(chan struct{}) + close(batch.ready) + cache[start] = batch + raw = nil + } + } + return cache +} + +func TestEntryDecodeReusesOnlyExactPrefetchedBytesAndBounds(t *testing.T) { + component, err := NewEntryBatch([]Entry{{Hash: solana.Hash{9}, Txns: []solana.Transaction{mustParseTransferTx(t, 21)}}}) + require.NoError(t, err) + raw, err := MarshalBlockComponent(component) + require.NoError(t, err) + for _, mode := range []string{"match", "different_bytes", "different_end", "different_start"} { + t.Run(mode, func(t *testing.T) { + shreds := paddedComponentShreds(t, raw, 0, 0xa5) + cache := decodedPrefetchForTest(shreds) + cached := cache[0] + require.NoError(t, cached.err) + cached.parseDuration = time.Hour // must not enter final parse accounting + if mode != "match" { + cached.err = errors.New("stale cached failure must be ignored") + } + switch mode { + case "different_bytes": + cached.raw[len(cached.raw)-1] ^= 1 // even differing padding invalidates reuse + case "different_end": + cached.end++ + cached.ready = make(chan struct{}) // mismatched bounds must never wait + case "different_start": + cached.start++ + cached.ready = make(chan struct{}) + } + timings := entryDecodeTimings{prefetched: cache} + entries, _, _, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.NoError(t, err) + require.Len(t, entries, 1) + require.Len(t, timings.all, 1) + require.Len(t, timings.retained, 1) + if mode == "match" { + require.Same(t, cached, timings.retained[0]) + require.Same(t, &cached.entries[0].Txns[0], &entries[0].Txns[0]) + require.Zero(t, timings.transactionParse) + } else { + require.NotSame(t, cached, timings.retained[0]) + require.Less(t, timings.transactionParse, time.Hour) + } + }) + } +} + +func TestEntryDecodePrefetchPreparationWaitHonorsCancellation(t *testing.T) { + raw, err := marshalEntryBatch([]Entry{{Hash: solana.Hash{3}}}) + require.NoError(t, err) + shreds := paddedComponentShreds(t, raw, 0, 0) + cache := decodedPrefetchForTest(shreds) + cache[0].ready = make(chan struct{}) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + timings := entryDecodeTimings{ctx: ctx, prefetched: cache} + _, _, _, err = decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.ErrorIs(t, err, context.Canceled) + require.Empty(t, timings.retained) + require.Zero(t, timings.transactionParse) +} + +func TestEntryDecodePrefetchPreservesUpdateParentAndIgnoresVerificationErrors(t *testing.T) { + prefix, err := NewEntryBatch([]Entry{{Hash: solana.Hash{1}, Txns: []solana.Transaction{mustParseTransferTx(t, 20)}}}) + require.NoError(t, err) + suffix, err := NewEntryBatch([]Entry{{Hash: solana.Hash{2}, Txns: []solana.Transaction{mustParseTransferTx(t, 21)}}}) + require.NoError(t, err) + components := []BlockComponent{NewBlockHeader(99, solana.Hash{9}), prefix, NewUpdateParent(98, solana.Hash{8}), suffix, NewBlockFooter(BlockFooter{BankHash: solana.Hash{7}})} + var shreds []*Shred + for _, component := range components { + raw, err := MarshalBlockComponent(component) + require.NoError(t, err) + shreds = append(shreds, paddedComponentShreds(t, raw, uint32(len(shreds)), 0xa5)...) + } + cache := decodedPrefetchForTest(shreds) + for _, start := range []uint32{32, 96} { + // Decoding must not make a signature decision, even for retained entries. + cache[start].verification = &transactionVerification{err: errors.New("signature result belongs to completion")} + cache[start].submitErr = errors.New("submission result belongs to completion") + } + timings := entryDecodeTimings{prefetched: cache} + entries, parent, footer, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.NoError(t, err) + require.Len(t, entries, 1) + require.Equal(t, solana.Hash{2}, entries[0].Hash) + require.Equal(t, uint32(64), parent.ReplayFECSetIndex) + require.Equal(t, solana.Hash{7}, footer.BankHash) + require.Len(t, timings.all, 5) + require.Len(t, timings.retained, 1) + require.Same(t, cache[96], timings.retained[0]) + + // Parsing the discarded prefix remains mandatory, unlike its signatures. + cache[32].err = errors.New("malformed optimistic prefix") + timings = entryDecodeTimings{prefetched: cache} + _, _, _, err = decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.ErrorContains(t, err, "malformed optimistic prefix") +} + +func TestEntryDecodePrefetchedAgavePaddedCaptureMatchesFullDecode(t *testing.T) { + packets := agavePaddedSlot1752420Packets(t) + shreds := make([]*Shred, len(packets)) + for i, packet := range packets { + var err error + shreds[i], err = ParseShred(packet) + require.NoError(t, err) + } + wantEntries, wantParent, wantFooter, err := DecodeEntriesAndAlpenglowMarkersFromDataShreds(shreds) + require.NoError(t, err) + timings := entryDecodeTimings{prefetched: decodedPrefetchForTest(shreds)} + entries, parent, footer, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.NoError(t, err) + require.Equal(t, wantEntries, entries) + require.Equal(t, wantParent, parent) + require.Equal(t, wantFooter, footer) + require.Len(t, timings.all, 4) + require.Len(t, timings.retained, 2) + require.Zero(t, timings.transactionParse) +} diff --git a/pkg/turbine/entry_batch_index.go b/pkg/turbine/entry_batch_index.go new file mode 100644 index 000000000..3bf3f2640 --- /dev/null +++ b/pkg/turbine/entry_batch_index.go @@ -0,0 +1,162 @@ +package turbine + +import "math/bits" + +// Three levels cover all 65,536 permitted data-shred indexes. Successor and +// predecessor queries touch a bounded number of words, even in adversarial order. +// The index is bounded (~24 KiB per retained slot) and allocated only when +// streaming preparation is enabled. It never rescans a slot's retained map. +const batchIndexWords = maxDataShredsPerSlot / 64 + +type shredIndexBits struct { + words [batchIndexWords]uint64 + groups [batchIndexWords / 64]uint64 + top uint64 +} + +func (b *shredIndexBits) set(i uint32) { + w, g := i/64, i/4096 + b.words[w] |= uint64(1) << (i % 64) + b.groups[g] |= uint64(1) << (w % 64) + b.top |= uint64(1) << g +} +func (b *shredIndexBits) clear(i uint32) { + w, g := i/64, i/4096 + b.words[w] &^= uint64(1) << (i % 64) + if b.words[w] == 0 { + b.groups[g] &^= uint64(1) << (w % 64) + if b.groups[g] == 0 { + b.top &^= uint64(1) << g + } + } +} +func (b *shredIndexBits) next(i uint32) (uint32, bool) { + if i >= maxDataShredsPerSlot { + return 0, false + } + w, g := i/64, i/4096 + if x := b.words[w] & (^uint64(0) << (i % 64)); x != 0 { + return w*64 + uint32(bits.TrailingZeros64(x)), true + } + x := b.groups[g] & (^uint64(0) << (w%64 + 1)) + if x == 0 { + top := b.top & (^uint64(0) << (g + 1)) + if top == 0 { + return 0, false + } + g = uint32(bits.TrailingZeros64(top)) + x = b.groups[g] + } + w = g*64 + uint32(bits.TrailingZeros64(x)) + return w*64 + uint32(bits.TrailingZeros64(b.words[w])), true +} +func (b *shredIndexBits) previous(i uint32) (uint32, bool) { + if i >= maxDataShredsPerSlot { + i = maxDataShredsPerSlot - 1 + } + w, g := i/64, i/4096 + if x := b.words[w] & (^uint64(0) >> (63 - i%64)); x != 0 { + return w*64 + uint32(63-bits.LeadingZeros64(x)), true + } + x := b.groups[g] & ((uint64(1) << (w % 64)) - 1) + if x == 0 { + top := b.top & ((uint64(1) << g) - 1) + if top == 0 { + return 0, false + } + g = uint32(63 - bits.LeadingZeros64(top)) + x = b.groups[g] + } + w = g*64 + uint32(63-bits.LeadingZeros64(x)) + return w*64 + uint32(63-bits.LeadingZeros64(b.words[w])), true +} + +type entryBatchIndex struct { + missing shredIndexBits + ends shredIndexBits + emitted [batchIndexWords]uint64 +} + +func newEntryBatchIndex() *entryBatchIndex { + b := new(entryBatchIndex) + for i := range b.missing.words { + b.missing.words[i] = ^uint64(0) + } + for i := range b.missing.groups { + b.missing.groups[i] = ^uint64(0) + } + b.missing.top = (uint64(1) << len(b.missing.groups)) - 1 + return b +} + +// One insertion can complete its containing batch, and, if it provides a new +// DATA_COMPLETE boundary, the immediately following batch. All other batches +// are unchanged. A range is emitted once, only when every data index is present. +// The preceding boundary is mandatory unless the range starts at index zero. +func (b *entryBatchIndex) add(i uint32, dataComplete bool) (ready [2]shredBatchRange, n int) { + if i >= maxDataShredsPerSlot { + return ready, 0 + } + b.missing.clear(i) + if dataComplete { + b.ends.set(i) + } + if end, ok := b.ends.next(i); ok { + if r, ok := b.complete(end); ok { + ready[n] = r + n++ + } + } + if dataComplete { + if end, ok := b.ends.next(i + 1); ok { + if r, ok := b.complete(end); ok { + ready[n] = r + n++ + } + } + } + return +} +func (b *entryBatchIndex) complete(end uint32) (shredBatchRange, bool) { + if b.emitted[end/64]&(uint64(1)<<(end%64)) != 0 { + return shredBatchRange{}, false + } + start := uint32(0) + if end > 0 { + if prev, ok := b.ends.previous(end - 1); ok { + start = prev + 1 + } + } + if missing, ok := b.missing.next(start); ok && missing <= end { + return shredBatchRange{}, false + } + b.emitted[end/64] |= uint64(1) << (end % 64) + return shredBatchRange{start, end}, true +} + +func (s *slotState) discoverEntryBatch(sh *Shred) { + ready, n := s.batchIndex.add(sh.Index, sh.DataComplete()) + for _, r := range ready[:n] { + s.completeBatches = append(s.completeBatches, r) + if s.pipelineTrace != nil && !s.pipelineTrace.sealed { + s.pipelineTrace.discovered[r.start] = entryTraceNow() + } + } +} + +// Seed once if preparation is installed on an assembler with existing data. +// Normal ingress creates the index with the first accepted data shred. +func (a *SlotAssembler) notePrefetchShredLocked(s *slotState, sh *Shred) { + p := a.entryPrefetch + if p == nil || p.closed || p.ctx.Err() != nil || sh.Type != ShredTypeData { + return + } + if s.batchIndex == nil { + s.batchIndex = newEntryBatchIndex() + for _, existing := range s.shreds { + s.discoverEntryBatch(existing) + } + } else { + s.discoverEntryBatch(sh) + } +} diff --git a/pkg/turbine/entry_batch_index_test.go b/pkg/turbine/entry_batch_index_test.go new file mode 100644 index 000000000..5f812be8a --- /dev/null +++ b/pkg/turbine/entry_batch_index_test.go @@ -0,0 +1,106 @@ +package turbine + +import ( + "github.com/stretchr/testify/require" + "math/rand" + "testing" +) + +func TestShredIndexBitsBoundaries(t *testing.T) { + var b shredIndexBits + indexes := []uint32{0, 1, 63, 64, 65, 4095, 4096, 4097, 65534, 65535} + for _, i := range indexes { + b.set(i) + } + for i := uint32(0); i <= 65536; i++ { + var wantNext, wantPrev uint32 + var hasNext, hasPrev bool + for _, j := range indexes { + if j >= i && !hasNext { + wantNext, hasNext = j, true + } + if j <= i { + wantPrev, hasPrev = j, true + } + } + got, ok := b.next(i) + require.Equal(t, hasNext, ok) + if ok { + require.Equal(t, wantNext, got) + } + got, ok = b.previous(i) + require.Equal(t, hasPrev, ok) + if ok { + require.Equal(t, wantPrev, got) + } + } + for _, i := range indexes { + b.clear(i) + } + _, ok := b.next(0) + require.False(t, ok) + _, ok = b.previous(65535) + require.False(t, ok) + require.Zero(t, b.top) +} + +func TestEntryBatchIndexRandomArrivalMatchesOracle(t *testing.T) { + rng := rand.New(rand.NewSource(42)) + for trial := 0; trial < 100; trial++ { + const size = 257 + var ends, present [size]bool + for i := range ends { + ends[i] = rng.Intn(8) == 0 + } + ends[size-1] = true + index := newEntryBatchIndex() + emitted := map[shredBatchRange]bool{} + for _, i := range rng.Perm(size) { + present[i] = true + got, n := index.add(uint32(i), ends[i]) + want := map[shredBatchRange]bool{} + start, complete := 0, true + for j := 0; j < size; j++ { + complete = complete && present[j] + if ends[j] && present[j] { + r := shredBatchRange{uint32(start), uint32(j)} + if complete && !emitted[r] { + want[r] = true + } + start, complete = j+1, true + } + } + require.Len(t, want, n, "trial %d index %d", trial, i) + for _, r := range got[:n] { + require.True(t, want[r]) + require.False(t, emitted[r]) + emitted[r] = true + } + _, n = index.add(uint32(i), ends[i]) + require.Zero(t, n, "duplicate emitted") + } + } +} + +func BenchmarkEntryBatchIndex(b *testing.B) { + for _, reverse := range []bool{false, true} { + name := "ordered" + if reverse { + name = "reverse" + } + b.Run(name, func(b *testing.B) { + b.ReportAllocs() + for n := 0; n < b.N; n++ { + idx := newEntryBatchIndex() + for k := uint32(0); k < 65536; k++ { + i := k + if reverse { + i = 65535 - k + } + idx.add(i, i%64 == 63) + } + } + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N)/65536, "ns/shred") + }) + } +} diff --git a/pkg/turbine/entry_batch_transactions_test.go b/pkg/turbine/entry_batch_transactions_test.go new file mode 100644 index 000000000..28329ab03 --- /dev/null +++ b/pkg/turbine/entry_batch_transactions_test.go @@ -0,0 +1,37 @@ +package turbine + +import ( + "fmt" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestEntryBatchTransactionsPreservesEntryOwnershipAndOrder(t *testing.T) { + entries := []Entry{{}, {Txns: make([]solana.Transaction, 3)}, {}, {Txns: make([]solana.Transaction, 2)}} + txs := entryBatchTransactions(entries) + require.Len(t, txs, 5) + for i := 0; i < 3; i++ { + require.Same(t, &entries[1].Txns[i], txs[i]) + } + for i := 0; i < 2; i++ { + require.Same(t, &entries[3].Txns[i], txs[3+i]) + } + require.Empty(t, entryBatchTransactions(nil)) + require.Empty(t, entryBatchTransactions([]Entry{{}, {}})) +} + +var entryBatchBenchmarkSink []*solana.Transaction + +func BenchmarkEntryBatchTransactions(b *testing.B) { + for _, count := range []int{4, 49, 269, 33760} { + b.Run(fmt.Sprintf("tx_%d", count), func(b *testing.B) { + entries := []Entry{{Txns: make([]solana.Transaction, count/2)}, {}, {Txns: make([]solana.Transaction, count-count/2)}} + b.ReportAllocs() + for b.Loop() { + entryBatchBenchmarkSink = entryBatchTransactions(entries) + } + }) + } +} diff --git a/pkg/turbine/entry_hash.go b/pkg/turbine/entry_hash.go index 72af6bdb8..35ee7561c 100644 --- a/pkg/turbine/entry_hash.go +++ b/pkg/turbine/entry_hash.go @@ -90,14 +90,7 @@ func hashTransactions(txns []solana.Transaction) solana.Hash { } func hashSignatures(signatures [][]byte) solana.Hash { - if len(signatures) == 0 { - return solana.Hash{} - } - nodes := merkletree.HashNodes(signatures) - if root := nodes.GetRoot(); root != nil { - return solana.Hash(*root) - } - return solana.Hash{} + return solana.Hash(merkletree.HashRoot(signatures)) } func sha256Hash(data []byte) solana.Hash { diff --git a/pkg/turbine/entry_hash_bench_test.go b/pkg/turbine/entry_hash_bench_test.go new file mode 100644 index 000000000..e8d7d1038 --- /dev/null +++ b/pkg/turbine/entry_hash_bench_test.go @@ -0,0 +1,24 @@ +package turbine + +import ( + "encoding/binary" + "fmt" + "testing" +) + +func BenchmarkEntrySignatureRoot(b *testing.B) { + for _, count := range []int{1, 64, 311, 512, 1024} { + b.Run(fmt.Sprint(count), func(b *testing.B) { + sigs := make([][]byte, count) + for i := range sigs { + sigs[i] = make([]byte, 64) + binary.LittleEndian.PutUint64(sigs[i], uint64(i)) + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + _ = hashSignatures(sigs) + } + }) + } +} diff --git a/pkg/turbine/entry_identity_recovery_test.go b/pkg/turbine/entry_identity_recovery_test.go new file mode 100644 index 000000000..aab96f794 --- /dev/null +++ b/pkg/turbine/entry_identity_recovery_test.go @@ -0,0 +1,131 @@ +package turbine + +import ( + "context" + "sync" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestEntryIdentityRecoveryVerifiesFinalTransactions(t *testing.T) { + for _, fault := range []string{"partial_identities", "wrong_pointer", "missing_range", "oversized_range", "nil_batch"} { + for _, invalidFinal := range []bool{false, true} { + name := fault + "/valid" + if invalidFinal { + name = fault + "/invalid" + } + t.Run(name, func(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + txs := verifierSignedTransactions(t, 3) + entries := []Entry{{Txns: []solana.Transaction{*txs[0], *txs[1], *txs[2]}}} + blk := &block.Block{Slot: 812, Transactions: entryBatchTransactions(entries)} + request, err := v.submitTransactions(context.Background(), blk.Transactions) + require.NoError(t, err) + _, err = request.wait() + require.NoError(t, err) + batches := []*prefetchedShredBatch{{entries: entries, verification: request}} + switch fault { + case "partial_identities": + request.identities = request.identities[:2] + case "wrong_pointer": + copyTx := *blk.Transactions[0] + blk.Transactions[0] = ©Tx + case "missing_range": + batches = nil + case "oversized_range": + batches[0].entries = append(batches[0].entries, Entry{Txns: []solana.Transaction{*txs[0]}}) + case "nil_batch": + batches = []*prefetchedShredBatch{nil} + } + if invalidFinal { + // The old request verified a different, valid transaction. + // A successful old verdict must not bless these final bytes. + copyTx := *blk.Transactions[0] + copyTx.Signatures = append([]solana.Signature(nil), copyTx.Signatures...) + copyTx.Signatures[0][0] ^= 1 + blk.Transactions[0] = ©Tx + } + err = verifyDecodedEntryBatches(context.Background(), blk, batches, v) + if invalidFinal { + require.ErrorContains(t, err, "failed signature verification") + return + } + require.NoError(t, err) + prepared, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + for i, tx := range blk.Transactions { + want, err := txstatus.IdentityForTransaction(tx) + require.NoError(t, err) + require.Equal(t, want, prepared.Identity(i)) + } + }) + } + } +} + +func TestEntryIdentityRecoveryPreservesCancellationAndVerifierFailure(t *testing.T) { + blk := &block.Block{Slot: 813, Transactions: verifierSignedTransactions(t, 2)} + v := newTransactionVerifier(2, 16, nil) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + require.ErrorIs(t, verifyDecodedEntryBatches(ctx, blk, nil, v), context.Canceled) + v.closeAndWait() + require.ErrorIs(t, verifyDecodedEntryBatches(context.Background(), blk, nil, v), errTransactionVerifierClosed) +} + +func TestEntryIdentityRecoveryJoinsCanceledReaders(t *testing.T) { + started := make(chan struct{}, 1) + release := make(chan struct{}) + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { + select { + case started <- struct{}{}: + default: + } + <-release + return nil + }) + defer v.closeAndWait() + var releaseOnce sync.Once + unblock := func() { releaseOnce.Do(func() { close(release) }) } + defer unblock() + txs := verifierSignedTransactions(t, 2) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + request, err := v.submitTransactions(ctx, txs) + require.NoError(t, err) + select { + case <-started: + case <-time.After(time.Second): + t.Fatal("reader never started") + } + result := make(chan error, 1) + // This deliberately inconsistent range sends us through recovery while + // the old verifier still owns the block's transaction buffers. + go func() { + result <- verifyDecodedEntryBatches(ctx, &block.Block{Transactions: txs}, []*prefetchedShredBatch{{verification: request}}, v) + }() + cancel() + select { + case err := <-result: + t.Fatalf("returned before old reader released its buffers: %v", err) + case <-time.After(20 * time.Millisecond): + } + unblock() + select { + case err := <-result: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(time.Second): + t.Fatal("recovery did not join canceled readers") + } + select { + case <-request.done: + default: + t.Fatal("request ownership was not released") + } +} diff --git a/pkg/turbine/entry_pipeline_trace.go b/pkg/turbine/entry_pipeline_trace.go new file mode 100644 index 000000000..38ea1055e --- /dev/null +++ b/pkg/turbine/entry_pipeline_trace.go @@ -0,0 +1,333 @@ +package turbine + +// Temporary, opt-in pipeline diagnostics. No transaction bytes or keys are logged. +// Timestamps are monotonic nanoseconds relative to origin_unix_ns. Worker elapsed +// time includes descheduling; summed job durations are NOT wall-clock critical paths. +import ( + "context" + "encoding/json" + "fmt" + "os" + "strconv" + "sync/atomic" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" +) + +var entryTraceOrigin = time.Now() +var entryTraceDropped atomic.Uint64 +var entryTraceConfig = configureEntryTrace() + +type entryTraceSettings struct { + modulo uint64 + until time.Time + reports chan entryPipelineReport + repairs chan entryRepairTrace +} + +func configureEntryTrace() entryTraceSettings { + n, err := strconv.ParseUint(os.Getenv("MITHRIL_ENTRY_TRACE_MOD"), 10, 32) + if err != nil || n == 0 { + return entryTraceSettings{} + } + seconds, err := strconv.Atoi(os.Getenv("MITHRIL_ENTRY_TRACE_SECONDS")) + if err != nil || seconds < 1 || seconds > 1800 { + seconds = 600 + } + path := os.Getenv("MITHRIL_ENTRY_TRACE_FILE") + if path == "" { + return entryTraceSettings{} + } + f, err := os.OpenFile(path, os.O_CREATE|os.O_APPEND|os.O_WRONLY, 0600) + if err != nil { + fmt.Fprintf(os.Stderr, "entry pipeline trace disabled: %v\n", err) + return entryTraceSettings{} + } + ch := make(chan entryPipelineReport, 8) + repairs := make(chan entryRepairTrace, 256) + go func() { + defer f.Close() + enc := json.NewEncoder(f) + for { + select { + case r := <-repairs: + r.Dropped = entryTraceDropped.Load() + if err := enc.Encode(r); err != nil { + entryTraceDropped.Add(1) + } + case r := <-ch: + r.Dropped = entryTraceDropped.Load() + r.finish() + if err := enc.Encode(r); err != nil { + entryTraceDropped.Add(1) + } + } + } + }() + return entryTraceSettings{modulo: n, until: time.Now().Add(time.Duration(seconds) * time.Second), reports: ch, repairs: repairs} +} + +func entryTraceNow() int64 { return time.Since(entryTraceOrigin).Nanoseconds() } +func entryTraceTime(t time.Time) int64 { return t.Sub(entryTraceOrigin).Nanoseconds() } + +type entryTraceContextKey struct{} + +func withEntryPipelineTrace(ctx context.Context, t *entryPipelineTrace) context.Context { + if t == nil { + return ctx + } + return context.WithValue(ctx, entryTraceContextKey{}, true) +} +func entryTraceContext(ctx context.Context) bool { + return ctx != nil && ctx.Value(entryTraceContextKey{}) == true +} + +type entryPipelineTrace struct { + arrivals map[uint32]int64 + sources map[uint32]entryShredSource + discovered map[uint32]int64 + sealed bool // frozen at first completion claim, including retry/error paths +} + +// Called only after successful admission, under the assembler lock. Includes +// FEC-reconstructed data. This is local availability, not a NIC timestamp. +func (s *slotState) traceAcceptedShred(sh *Shred, source ...entryShredSource) { + if sh.Type != ShredTypeData { + return + } + if s.pipelineTrace == nil { + c := entryTraceConfig + if c.modulo == 0 || s.slot%c.modulo != 0 || time.Now().After(c.until) { + return + } + s.pipelineTrace = &entryPipelineTrace{arrivals: make(map[uint32]int64), discovered: make(map[uint32]int64)} + } + if !s.pipelineTrace.sealed { + if _, exists := s.pipelineTrace.arrivals[sh.Index]; exists { + return + } + s.pipelineTrace.arrivals[sh.Index] = entryTraceNow() + if len(source) > 0 { + if s.pipelineTrace.sources == nil { + s.pipelineTrace.sources = make(map[uint32]entryShredSource) + } + s.pipelineTrace.sources[sh.Index] = source[0] + } + } +} + +type entryVerificationTrace struct { + Transactions int `json:"transactions"` + Submit int64 `json:"submit_ns"` + Admitted int64 `json:"admitted_ns"` + FirstWorker int64 `json:"first_worker_ns"` + LastWorker int64 `json:"last_worker_ns"` + Finished int64 `json:"finished_ns"` + Jobs int `json:"jobs"` + JobWaitSum int64 `json:"job_offer_to_start_sum_ns"` + JobWaitMax int64 `json:"job_offer_to_start_max_ns"` + WorkerSum int64 `json:"worker_elapsed_sum_ns"` + WorkerMax int64 `json:"worker_elapsed_max_ns"` +} + +func (t *entryVerificationTrace) observe(j *transactionVerifyJob) { + if t.FirstWorker == 0 || j.workerStart < t.FirstWorker { + t.FirstWorker = j.workerStart + } + t.LastWorker = max(t.LastWorker, j.workerEnd) + t.Jobs++ + wait, work := j.workerStart-j.offeredAt, j.workerEnd-j.workerStart + t.JobWaitSum += wait + t.JobWaitMax = max(t.JobWaitMax, wait) + t.WorkerSum += work + t.WorkerMax = max(t.WorkerMax, work) +} + +type entryBatchTraceReport struct { + CriticalIndex *uint32 `json:"critical_shred_index,omitempty"` + CriticalSource *entryShredSource `json:"critical_shred_source,omitempty"` + CriticalTies int `json:"critical_timestamp_ties"` + Start uint32 `json:"start"` + End uint32 `json:"end"` + Transactions int `json:"transactions"` + Retained bool `json:"retained"` + Prefetched bool `json:"prefetched"` + AvailabilityKnown bool `json:"availability_known"` + Available int64 `json:"available_ns"` + Discovered int64 `json:"discovered_ns"` + DecodeStart int64 `json:"decode_start_ns"` + DecodeEnd int64 `json:"decode_end_ns"` + Verification *entryVerificationTrace `json:"verification,omitempty"` +} + +type entryPipelineReport struct { + Origin int64 `json:"origin_unix_ns"` + Slot uint64 `json:"slot"` + Transactions int `json:"transactions"` + Full int64 `json:"full_ns"` + CompletionStart int64 `json:"completion_start_ns"` + Ready int64 `json:"ready_ns"` + Dropped uint64 `json:"dropped_reports"` + Batches []entryBatchTraceReport `json:"batches"` + Fallback *entryVerificationTrace `json:"fallback,omitempty"` + source *entryPipelineTrace + all, retained []*prefetchedShredBatch + fallback *transactionVerification +} + +func completedEntryVerification(r *transactionVerification) *entryVerificationTrace { + if r == nil { + return nil + } + select { + case <-r.done: + return r.trace + default: + return nil + } +} + +// All source maps are sealed, and decode has joined all preparation readers. +// Reading request metrics additionally requires the verification done barrier. +func (r *entryPipelineReport) finish() { + retained := make(map[*prefetchedShredBatch]bool, len(r.retained)) + for _, b := range r.retained { + retained[b] = true + } + for _, b := range r.all { + row := entryBatchTraceReport{Start: b.start, End: b.end, Retained: retained[b], Prefetched: b.ready != nil, + DecodeStart: b.traceDecodeStart, DecodeEnd: b.traceDecodeEnd, Discovered: r.source.discovered[b.start], + AvailabilityKnown: true, Verification: completedEntryVerification(b.verification)} + for _, e := range b.entries { + row.Transactions += len(e.Txns) + } + // Require the preceding DATA_COMPLETE boundary as well as every shred in + // this batch. An end marker alone cannot establish an independent start. + start := b.start + if start > 0 { + start-- + } + for i := start; i <= b.end; i++ { + at, ok := r.source.arrivals[i] + if !ok { + row.AvailabilityKnown = false + } + if ok && (row.CriticalIndex == nil || at > row.Available) { + index := i + row.CriticalIndex = &index + row.CriticalTies = 1 + row.CriticalSource = nil + if source, exists := r.source.sources[i]; exists { + row.CriticalSource = &source + } + row.Available = at + } else if ok && at == row.Available { + row.CriticalTies++ + } + } + r.Batches = append(r.Batches, row) + } + r.Fallback = completedEntryVerification(r.fallback) +} + +func queueEntryPipelineReport(s *slotState, b *block.Block, d *entryDecodeTimings, start, ready time.Time) { + if s.pipelineTrace == nil || len(b.Transactions) < 10000 || entryTraceConfig.reports == nil { + return + } + r := entryPipelineReport{Origin: entryTraceOrigin.UnixNano(), Slot: s.slot, Transactions: len(b.Transactions), + Full: entryTraceTime(s.fullAt), CompletionStart: entryTraceTime(start), Ready: entryTraceTime(ready), + source: s.pipelineTrace, all: d.all, retained: d.retained, fallback: d.traceFallback} + select { + case entryTraceConfig.reports <- r: + default: + entryTraceDropped.Add(1) + } +} + +// Attribution starts at assembler entry, not socket receipt. +// "non_repair" includes direct/spooled admission, not proof of socket origin. +// A recovered shred records the packet that triggered reconstruction; it is not itself a +// received repair response. Missing source fields in older reports mean unknown. +type entryShredSource struct { + Path string `json:"path"` + FEC uint32 `json:"fec_set"` + TriggerIndex uint32 `json:"trigger_index"` + TriggerCoding bool `json:"trigger_coding"` + TriggerRepair bool `json:"trigger_repair"` + AdmissionEntered int64 `json:"admission_entered_ns"` +} + +// Separate JSONL records join by origin/slot/index. The time pair brackets the +// UDP write syscall, NOT delivery. Attempt IDs can reset; order by timestamps. +// Highest-index probes are not exact requests for the returned shred index. +type entryRepairTrace struct { + AdmissionStart int64 `json:"admission_start_ns,omitempty"` + Peer string `json:"peer,omitempty"` + Nonce uint32 `json:"nonce"` + ResponseAt int64 `json:"response_ns,omitempty"` + RequestedAt int64 `json:"requested_ns,omitempty"` + ReturnedIndex uint32 `json:"returned_index"` + Late bool `json:"late"` + FEC uint32 `json:"fec_set"` + DeficitBefore int `json:"deficit_before"` + DeficitAfter int `json:"deficit_after"` + Recovered int `json:"recovered"` + Outcome string `json:"outcome,omitempty"` + Event string `json:"event"` + Origin int64 `json:"origin_unix_ns"` + Slot uint64 `json:"slot"` + Index uint32 `json:"index"` + Highest bool `json:"highest_index_probe"` + Attempt uint8 `json:"attempt"` + SendStart int64 `json:"send_start_ns"` + SendEnd int64 `json:"send_end_ns"` + Success bool `json:"success"` + Dropped uint64 `json:"dropped_reports"` +} + +func entryTraceSelected(slot uint64) bool { + c := entryTraceConfig + return c.modulo != 0 && slot%c.modulo == 0 && time.Now().Before(c.until) +} +func traceRepairSend(slot uint64, index uint32, kind repairRequestKind, attempt uint8, start int64, success bool, binding ...entryRepairTrace) { + if start == 0 || entryTraceConfig.repairs == nil { + return + } + r := entryRepairTrace{Event: "repair_send", Origin: entryTraceOrigin.UnixNano(), Slot: slot, Index: index, Highest: kind == repairRequestHighestWindowIndex, Attempt: attempt, SendStart: start, SendEnd: entryTraceNow(), Success: success} + if len(binding) > 0 { + r.Peer = binding[0].Peer + r.Nonce = binding[0].Nonce + } + emitRepairTrace(r) +} + +func emitRepairTrace(r entryRepairTrace) { + if entryTraceConfig.repairs == nil { + return + } + select { + case entryTraceConfig.repairs <- r: + default: + entryTraceDropped.Add(1) + } +} + +// -1 means no authenticated coding layout is known. Zero means sufficient +// shards, not that reconstruction necessarily succeeded (see outcome/recovered). +func traceFECDeficit(s *slotState, index uint32) int { + if s == nil { + return -1 + } + f := s.fecSets[index] + if f == nil || !f.haveLayout { + return -1 + } + return max(0, int(f.layout.dataShreds)-len(f.data)-len(f.coding)) +} +func traceRepairResponse(rec outstandingRepairRequest, sh *Shred, peer string, late bool) { + if !entryTraceSelected(sh.Slot) { + return + } + emitRepairTrace(entryRepairTrace{Event: "repair_response", Origin: entryTraceOrigin.UnixNano(), Slot: sh.Slot, Index: rec.key.index, Highest: rec.key.kind == repairRequestHighestWindowIndex, Attempt: rec.key.attempt, Nonce: rec.nonce, Peer: peer, RequestedAt: entryTraceTime(rec.sentAt), ResponseAt: entryTraceNow(), ReturnedIndex: sh.Index, FEC: sh.FECSetIndex, Late: late}) +} diff --git a/pkg/turbine/entry_pipeline_trace_test.go b/pkg/turbine/entry_pipeline_trace_test.go new file mode 100644 index 000000000..5816287d4 --- /dev/null +++ b/pkg/turbine/entry_pipeline_trace_test.go @@ -0,0 +1,177 @@ +package turbine + +import ( + "context" + "net" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestEntryPipelineTraceRequiresCompleteBatchAndBoundary(t *testing.T) { + source := &entryPipelineTrace{arrivals: map[uint32]int64{0: 10, 1: 100, 2: 20, 3: 30, 4: 40}, discovered: map[uint32]int64{0: 110, 3: 120}, sealed: true} + first := &prefetchedShredBatch{start: 0, end: 2} + later := &prefetchedShredBatch{start: 3, end: 4} + r := entryPipelineReport{source: source, all: []*prefetchedShredBatch{first, later}, retained: []*prefetchedShredBatch{later}} + r.finish() + require.Equal(t, int64(100), r.Batches[0].Available) + require.Equal(t, int64(40), r.Batches[1].Available) + require.Equal(t, int64(80), r.Batches[1].Discovered-r.Batches[1].Available, "a gap in the earlier batch delays discovery, not availability of the later one") + require.False(t, r.Batches[0].Retained) + require.True(t, r.Batches[1].Retained) + delete(source.arrivals, 2) + r.Batches = nil + r.finish() + require.False(t, r.Batches[1].AvailabilityKnown, "the preceding boundary must also have been observed") +} + +func TestEntryVerificationTraceJoinsWorkerTimings(t *testing.T) { + entered, release := make(chan struct{}), make(chan struct{}) + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { + select { + case <-entered: + default: + close(entered) + } + <-release + return nil + }) + defer v.closeAndWait() + ctx := withEntryPipelineTrace(context.Background(), &entryPipelineTrace{}) + r, err := v.submitTransactions(ctx, []*solana.Transaction{{}}) + require.NoError(t, err) + <-entered + require.Nil(t, completedEntryVerification(r), "unfinished mutable metrics must not be read") + close(release) + _, err = r.wait() + require.NoError(t, err) + m := completedEntryVerification(r) + require.NotNil(t, m) + require.Equal(t, 1, m.Jobs) + require.LessOrEqual(t, m.Submit, m.Admitted) + require.LessOrEqual(t, m.Admitted, m.FirstWorker) + require.Less(t, m.FirstWorker, m.LastWorker) + require.LessOrEqual(t, m.LastWorker, m.Finished) + require.Positive(t, m.WorkerSum) + require.GreaterOrEqual(t, m.JobWaitSum, int64(0)) + plain, err := v.submitTransactions(context.Background(), []*solana.Transaction{{}}) + require.NoError(t, err) + _, err = plain.wait() + require.NoError(t, err) + require.Nil(t, plain.trace, "ordinary verification does not collect job timestamps") +} + +func TestEntryPipelineTraceSealedGeneration(t *testing.T) { + s := &slotState{pipelineTrace: &entryPipelineTrace{arrivals: map[uint32]int64{1: 42}, discovered: map[uint32]int64{}, sealed: true}} + s.traceAcceptedShred(&Shred{Type: ShredTypeData, Index: 2}) + require.Len(t, s.pipelineTrace.arrivals, 1, "completion/retry cannot mutate a report's frozen arrival map") +} + +func TestEntryCriticalShredIncludesBoundaryAndSource(t *testing.T) { + source := &entryPipelineTrace{arrivals: map[uint32]int64{2: 100, 3: 20, 4: 30}, sources: map[uint32]entryShredSource{2: {Path: "fec_recovery", FEC: 0, TriggerIndex: 12, TriggerCoding: true, TriggerRepair: true, AdmissionEntered: 80}}} + r := entryPipelineReport{source: source, all: []*prefetchedShredBatch{{start: 3, end: 4}}} + r.finish() + require.Equal(t, uint32(2), *r.Batches[0].CriticalIndex) + require.Equal(t, "fec_recovery", r.Batches[0].CriticalSource.Path) + require.True(t, r.Batches[0].CriticalSource.TriggerRepair) + require.Equal(t, 1, r.Batches[0].CriticalTies) + source.arrivals[3] = 100 + r.Batches = nil + r.finish() + require.Equal(t, 2, r.Batches[0].CriticalTies) + require.Equal(t, uint32(2), *r.Batches[0].CriticalIndex, "ties select the lowest index deterministically") +} + +func TestEntryTracePreservesFirstAdmission(t *testing.T) { + s := &slotState{pipelineTrace: &entryPipelineTrace{arrivals: make(map[uint32]int64)}} + sh := &Shred{Type: ShredTypeData, Index: 1} + s.traceAcceptedShred(sh, entryShredSource{Path: "repair"}) + first := s.pipelineTrace.arrivals[1] + s.traceAcceptedShred(sh, entryShredSource{Path: "non_repair"}) + require.Equal(t, first, s.pipelineTrace.arrivals[1]) + require.Equal(t, "repair", s.pipelineTrace.sources[1].Path) + s.pipelineTrace.sealed = true + s.traceAcceptedShred(&Shred{Type: ShredTypeData, Index: 2}, entryShredSource{Path: "repair"}) + require.Len(t, s.pipelineTrace.sources, 1) +} + +func TestEntryRepairTraceBoundedAndExplicit(t *testing.T) { + old := entryTraceConfig + defer func() { entryTraceConfig = old }() + entryTraceConfig = entryTraceSettings{repairs: make(chan entryRepairTrace, 1)} + traceRepairSend(15, 10, repairRequestWindowIndex, 2, 100, true) + r := <-entryTraceConfig.repairs + require.Equal(t, "repair_send", r.Event) + require.Equal(t, uint64(15), r.Slot) + require.False(t, r.Highest) + require.True(t, r.Success) + require.Equal(t, uint8(2), r.Attempt) + traceRepairSend(15, 10, repairRequestHighestWindowIndex, 0, 100, false) + dropped := entryTraceDropped.Load() + traceRepairSend(15, 11, repairRequestWindowIndex, 0, 100, true) + require.Equal(t, dropped+1, entryTraceDropped.Load(), "full diagnostic queue never blocks repair") + r = <-entryTraceConfig.repairs + require.True(t, r.Highest) + require.False(t, r.Success) + traceRepairSend(15, 1, repairRequestWindowIndex, 0, 0, true) + require.Empty(t, entryTraceConfig.repairs, "unsampled sends are ignored") +} + +func TestEntryTraceSelectionIsBounded(t *testing.T) { + old := entryTraceConfig + defer func() { entryTraceConfig = old }() + entryTraceConfig = entryTraceSettings{modulo: 5, until: time.Now().Add(time.Minute)} + require.True(t, entryTraceSelected(15)) + require.False(t, entryTraceSelected(16)) + entryTraceConfig.until = time.Now().Add(-time.Second) + require.False(t, entryTraceSelected(15)) + entryTraceConfig = entryTraceSettings{} + require.False(t, entryTraceSelected(15)) +} + +func TestEntryRepairResponseCorrelation(t *testing.T) { + old := entryTraceConfig + defer func() { entryTraceConfig = old }() + entryTraceConfig = entryTraceSettings{modulo: 1, until: time.Now().Add(time.Minute), repairs: make(chan entryRepairTrace, 8)} + from := &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1), Port: 8000} + addr, _ := repairAddressKeyFromUDP(from) + for _, late := range []bool{false, true} { + c := newPacingTestClient(t) + key := repairRequestKey{kind: repairRequestWindowIndex, slot: 42, index: 3} + rec := outstandingRepairRequest{key: key, nonce: 777, addr: addr, sentAt: time.Now().Add(-time.Second), accountAt: time.Now().Add(time.Second)} + responseKey := repairResponseKey{addr: addr, nonce: 777} + if late { + c.expiredCur[responseKey] = rec + } else { + c.outstanding[key] = rec + c.byResponse[responseKey] = key + } + require.False(t, observeRepairForTest(c, nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 4, Type: ShredTypeData})) + require.Empty(t, entryTraceConfig.repairs, "wrong index cannot be reported as matched") + require.True(t, observeRepairForTest(c, nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 3, Type: ShredTypeData})) + r := <-entryTraceConfig.repairs + require.Equal(t, "repair_response", r.Event) + require.Equal(t, uint32(777), r.Nonce) + require.Equal(t, from.String(), r.Peer) + require.Equal(t, late, r.Late) + require.Equal(t, uint32(3), r.ReturnedIndex) + require.Greater(t, r.ResponseAt, r.RequestedAt) + require.False(t, observeRepairForTest(c, nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 3, Type: ShredTypeData})) + require.Empty(t, entryTraceConfig.repairs, "consumed nonce cannot count twice") + } +} + +func TestEntryFECDeficit(t *testing.T) { + s := newRepairSelectionSlot(42) + require.Equal(t, -1, traceFECDeficit(s, 0)) + addCodedSet(s, 0, 32, 32, seq(0, 19), 8) + require.Equal(t, 4, traceFECDeficit(s, 0)) + s.fecSets[0].data[20] = &Shred{} + require.Equal(t, 3, traceFECDeficit(s, 0)) + for i := uint32(21); i < 32; i++ { + s.fecSets[0].data[i] = &Shred{} + } + require.Zero(t, traceFECDeficit(s, 0), "enough shards is not a negative deficit") +} diff --git a/pkg/turbine/entry_prefetch.go b/pkg/turbine/entry_prefetch.go new file mode 100644 index 000000000..28076cfb3 --- /dev/null +++ b/pkg/turbine/entry_prefetch.go @@ -0,0 +1,479 @@ +package turbine + +import ( + "context" + "errors" + "sort" + "sync" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +const ( + entryPrefetchSlots = 8 + // Covers a large block of maximum-size transactions while bounding the + // extra retained encoded bytes across all in-flight generations. + entryPrefetchBytes = 64 << 20 + entryPrefetchBatchBytes = 1 << 20 +) + +type shredBatchRange struct{ start, end uint32 } + +// A result belongs to one exact DATA_COMPLETE range in one slot generation. +// Fields are immutable after ready closes; signature readers own its decoded +// transactions until verification.done closes. +type prefetchedShredBatch struct { + viewOnce sync.Once // protects the immutable stream view, including concurrent Resolve + view *StreamBatch + readyAt time.Time // set before ready closes + start, end uint32 + raw []byte + entries []Entry + transactions []*solana.Transaction // immutable pointer view, built once before ready closes + parent *AlpenglowParentInfo + footer *BlockFooter + marker bool + traceDecodeStart int64 + traceDecodeEnd int64 + parseDuration time.Duration + err error + ready chan struct{} + verification *transactionVerification + submittedAt time.Time + submitErr error +} + +type slotEntryPrefetch struct { + pool *entryPrefetchPool + ctx context.Context + cancel context.CancelFunc + batches map[uint32]*prefetchedShredBatch + next int + queued, released bool + queueDone chan struct{} // closed after the queued/running token retires + bytes int + budgetBlocked bool +} + +// All scheduling and accounting use assembler.mu. The packet reader only +// indexes complete data ranges and attempts a nonblocking, coalesced enqueue. +// Decoding and bounded verifier admission run on separate background workers. +type entryPrefetchPool struct { + a *SlotAssembler + ctx context.Context + cancel context.CancelFunc + verifier *transactionVerifier + jobs chan *slotState + workers, cleanup sync.WaitGroup + slots, bytes int + active map[*slotState]struct{} // at most entryPrefetchSlots admitted generations + closed bool + close sync.Once +} + +func newEntryPrefetchPool(ctx context.Context, a *SlotAssembler, verifier *transactionVerifier) *entryPrefetchPool { + ctx, cancel := context.WithCancel(ctx) + p := &entryPrefetchPool{a: a, ctx: ctx, cancel: cancel, verifier: verifier, jobs: make(chan *slotState, entryPrefetchSlots)} + a.mu.Lock() + a.entryPrefetch = p + a.mu.Unlock() + p.workers.Add(2) + for i := 0; i < 2; i++ { + go p.run() + } + return p +} + +func (a *SlotAssembler) prefetchEntriesLocked(s *slotState) { + p := a.entryPrefetch + if p == nil || p.closed || p.ctx.Err() != nil || s.streamCancelReason != "" { + return + } + if s.batchIndex == nil && len(s.shreds) != 0 { + s.batchIndex = newEntryBatchIndex() + for _, sh := range s.shreds { + s.discoverEntryBatch(sh) + } + } + if s.prefetch == nil && len(s.completeBatches) > 0 && p.slots < entryPrefetchSlots { + ctx, cancel := context.WithCancel(withEntryPipelineTrace(p.ctx, s.pipelineTrace)) + s.prefetch = &slotEntryPrefetch{pool: p, ctx: ctx, cancel: cancel, batches: make(map[uint32]*prefetchedShredBatch)} + p.slots++ + if p.active == nil { + p.active = make(map[*slotState]struct{}) + } + p.active[s] = struct{}{} + } + p.enqueueLocked(s) +} + +func (p *entryPrefetchPool) enqueueLocked(s *slotState) { + f := s.prefetch + if f == nil || f.pool != p || f.queued || f.released || s.completing || p.closed || f.next >= len(s.completeBatches) { + return + } + select { + case p.jobs <- s: + f.queued = true + f.queueDone = make(chan struct{}) + default: + } +} + +func (p *entryPrefetchPool) run() { + defer p.workers.Done() + for s := range p.jobs { + p.a.mu.Lock() + f := s.prefetch + if p.closed || f.released || f.ctx.Err() != nil || p.a.slots[s.slot] != s || s.completing { + f.queued = false + close(f.queueDone) + p.a.mu.Unlock() + continue + } + f.budgetBlocked = false + var batch *prefetchedShredBatch + var shreds []*Shred + var rawSize int + for f.next < len(s.completeBatches) { + r := s.completeBatches[f.next] + size := 0 + for i := r.start; i <= r.end; i++ { + size += len(s.shreds[i].Data) + if size > entryPrefetchBatchBytes { + break + } + } + if size > entryPrefetchBatchBytes { + // Never publish a prefix with an unfillable hole. Completion + // still decodes and verifies the entire valid block normally. + s.streamCancelReason = "prefetch_batch_too_large" + p.a.releasePrefetchLocked(s) + break + } + if p.bytes+size > entryPrefetchBytes { + f.budgetBlocked = true + break + } + f.next++ + p.bytes += size + f.bytes += size + rawSize = size + batch = &prefetchedShredBatch{start: r.start, end: r.end, ready: make(chan struct{})} + f.batches[r.start] = batch + shreds = make([]*Shred, 0, r.end-r.start+1) + for i := r.start; i <= r.end; i++ { + shreds = append(shreds, s.shreds[i]) + } + break + } + if batch == nil { + f.queued = false + close(f.queueDone) + p.a.mu.Unlock() + continue + } + p.a.mu.Unlock() + + if s.pipelineTrace != nil { + batch.traceDecodeStart = entryTraceNow() + } + raw := make([]byte, 0, rawSize) + for _, sh := range shreds { + raw = append(raw, sh.Data...) + } + decoded := decodeClosedShredBatch(raw, batch.start, batch.end) + ready := batch.ready + batch.raw, batch.entries = decoded.raw, decoded.entries + batch.parent, batch.footer, batch.marker = decoded.parent, decoded.footer, decoded.marker + batch.parseDuration, batch.err = decoded.parseDuration, decoded.err + if s.pipelineTrace != nil { + batch.traceDecodeEnd = entryTraceNow() + } + if batch.err == nil && !batch.marker { + batch.transactions = entryBatchTransactions(batch.entries) + } + if batch.err == nil && !batch.marker && f.ctx.Err() == nil { + txs := batch.transactions + if len(txs) > 0 { + batch.submittedAt = time.Now() + batch.verification, batch.submitErr = p.verifier.submitPrefetchTransactions(f.ctx, txs) + } + } + batch.readyAt = time.Now() + close(ready) + p.a.mu.Lock() + f.queued = false + close(f.queueDone) + if !f.released && p.a.slots[s.slot] == s { + p.a.publishStreamBatchReadyLocked(s, batch) + } + p.enqueueLocked(s) + p.a.mu.Unlock() + } +} + +func entryBatchTransactions(entries []Entry) []*solana.Transaction { + count := 0 + for i := range entries { + count += len(entries[i].Txns) + } + // Entries are already decoded: size this pointer view once instead of + // repeatedly reallocating and copying it while preparing each component. + txs := make([]*solana.Transaction, 0, count) + for i := range entries { + for j := range entries[i].Txns { + txs = append(txs, &entries[i].Txns[j]) + } + } + return txs +} + +// Keep reservations until canceled readers have actually relinquished their +// buffers. Repeated resets cannot evade the memory or slot bounds. +func (a *SlotAssembler) releasePrefetchLocked(s *slotState) { + if s == nil || s.prefetch == nil || s.prefetch.released { + return + } + f := s.prefetch + f.released = true + f.cancel() + reason := s.streamCancelReason + if reason == "" { + reason = "released" + } + a.publishStreamReleaseLocked(s, reason) + p := f.pool + queueDone := f.queueDone + p.cleanup.Add(1) + go func() { + defer p.cleanup.Done() + for _, b := range f.batches { + <-b.ready + if b.verification != nil { + b.verification.wait() + } + } + if queueDone != nil { + <-queueDone + } + p.a.mu.Lock() + p.slots-- + p.bytes -= f.bytes + delete(p.active, s) + p.retryBudgetBlockedLocked() + p.a.mu.Unlock() + }() +} + +// Retry only admitted generations, oldest slot first, when readers release bytes. +// This avoids both waiting for another shred and scanning all retained slots. +func (p *entryPrefetchPool) retryBudgetBlockedLocked() { + if p.closed || p.ctx.Err() != nil { + return + } + waiting := make([]*slotState, 0, len(p.active)) + for s := range p.active { + if s.prefetch.budgetBlocked && p.a.slots[s.slot] == s { + waiting = append(waiting, s) + } + } + sort.Slice(waiting, func(i, j int) bool { return waiting[i].slot < waiting[j].slot }) + for _, s := range waiting { + p.enqueueLocked(s) + } +} + +func (p *entryPrefetchPool) closeAndWait() { + p.close.Do(func() { + p.cancel() + p.a.mu.Lock() + p.closed = true + if p.a.entryPrefetch == p { + p.a.entryPrefetch = nil + } + for _, s := range p.a.slots { + if s.prefetch != nil && s.prefetch.pool == p { + if s.streamCancelReason == "" { + s.streamCancelReason = "shutdown" + } + p.a.releasePrefetchLocked(s) + } + } + close(p.jobs) + p.a.mu.Unlock() + p.workers.Wait() + p.cleanup.Wait() + }) +} + +// Reuse only retained entry results. UpdateParent may intentionally discard an +// invalid optimistic prefix. Unprefetched transactions form one immediately +// available request, overlapping any early requests still running. +func verifyDecodedEntryBatches(ctx context.Context, blk *block.Block, batches []*prefetchedShredBatch, verifier *transactionVerifier) error { + return verifyDecodedEntryBatchesWithTimings(ctx, blk, batches, verifier, nil) +} + +func verifyDecodedEntryBatchesWithTimings(ctx context.Context, blk *block.Block, batches []*prefetchedShredBatch, verifier *transactionVerifier, timings *entryDecodeTimings) error { + if ctx == nil { + ctx = context.Background() + } + if blk == nil { + return errors.New("verify decoded entries: nil block") + } + type pending struct { + future *transactionVerification + offset int + count int + } + var early []pending + var missing []*solana.Transaction + var indices []int + offset := 0 + for _, b := range batches { + if b == nil { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("nil retained entry batch")) + } + count := 0 + for _, e := range b.entries { + count += len(e.Txns) + } + if count > len(blk.Transactions)-offset { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("entry transaction range exceeds final block")) + } + reusable := b.verification != nil + if reusable { + select { + case <-b.verification.done: + // Cancellation is not a signature verdict. A completion canceled + // after preparation may be retried on this same slot generation. + reusable = !errors.Is(b.verification.err, context.Canceled) && !errors.Is(b.verification.err, context.DeadlineExceeded) + default: + } + } + if reusable { + early = append(early, pending{b.verification, offset, count}) + } else { + missing = append(missing, blk.Transactions[offset:offset+count]...) + for i := 0; i < count; i++ { + indices = append(indices, offset+i) + } + } + offset += count + } + if offset != len(blk.Transactions) { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("entry identity coverage mismatch")) + } + var fallback *transactionVerification + var err error + if len(missing) > 0 { + fallback, err = verifier.submitTransactions(ctx, missing) + if timings != nil { + timings.traceFallback = fallback + } + } + firstIndex := len(blk.Transactions) + firstErr := err + for _, p := range early { + i, e := p.future.waitContext(ctx) + if e != nil && (firstErr == nil || (i >= 0 && p.offset+i < firstIndex)) { + firstErr = e + if i >= 0 { + firstIndex = p.offset + i + } + } + } + if fallback != nil { + i, e := fallback.waitContext(ctx) + if e != nil && (firstErr == nil || (i >= 0 && indices[i] < firstIndex)) { + firstErr = e + if i >= 0 { + firstIndex = indices[i] + } + } + } + if ctx.Err() != nil { + return ctx.Err() + } + if firstErr != nil && firstIndex < len(blk.Transactions) { + return formatTransactionVerificationError(blk, firstIndex, firstErr) + } + if firstErr != nil { + return firstErr + } + // Custom verification hooks do not produce trusted message identities. + // Preserve their existing lazy preparation path (primarily test fixtures). + if verifier.verify != nil { + return nil + } + identities := make([]txverify.VerifiedMessageIdentity, len(blk.Transactions)) + for _, p := range early { + if len(p.future.identities) != p.count || p.offset+len(p.future.identities) > len(identities) { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("entry identity range mismatch")) + } + copy(identities[p.offset:], p.future.identities) + } + if fallback != nil { + if len(fallback.identities) != len(indices) { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("fallback identity coverage mismatch")) + } + for i, index := range indices { + identities[index] = fallback.identities[i] + } + } + if err := blk.CacheVerifiedTransactionMessageIdentities(identities); err != nil { + return recoverEntryVerification(ctx, blk, batches, verifier, err) + } + return nil +} + +// Prefetch metadata is an optimization, never a substitute for verifying the +// final block. Join old readers, then verify every final transaction afresh. +// This also repairs pointer/coverage mismatches without treating a successful +// verdict for another transaction as proof for this one. Normal signature +// failures above are still rejected directly. Failed recovery stays an error. +func recoverEntryVerification(ctx context.Context, blk *block.Block, batches []*prefetchedShredBatch, verifier *transactionVerifier, reason error) error { + for _, batch := range batches { + if batch != nil && batch.verification != nil { + _, _ = batch.verification.waitContext(ctx) + } + } + if err := ctx.Err(); err != nil { + return err + } + mlog.Log.Warnf("slot %d: discarded inconsistent entry verification metadata; re-verifying final transactions: %v", blk.Slot, reason) + return verifier.verifyBlockContext(ctx, blk) +} + +func earlyEntryTimings(t *entryDecodeTimings, fullAt time.Time, timings *block.TurbineIngressTimings) { + for _, b := range t.all { + if b.ready == nil { + continue + } + timings.EarlyTransactionParse += b.parseDuration + if b.verification != nil { + select { + case <-b.verification.done: + timings.EarlyTransactionSigverify += b.verification.finishedAt.Sub(b.submittedAt) + default: + } + } + } + for _, b := range t.retained { + if b.verification != nil { + select { + case <-b.verification.done: + if b.verification.err == nil && !b.verification.finishedAt.After(fullAt) { + for _, e := range b.entries { + timings.EarlyVerifiedTransactions += uint64(len(e.Txns)) + } + } + default: + } + } + } +} diff --git a/pkg/turbine/entry_prefetch_benchmark_test.go b/pkg/turbine/entry_prefetch_benchmark_test.go new file mode 100644 index 000000000..be1ef6dcd --- /dev/null +++ b/pkg/turbine/entry_prefetch_benchmark_test.go @@ -0,0 +1,288 @@ +package turbine + +import ( + "context" + "crypto/ed25519" + "encoding/binary" + "fmt" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/sigverify" + "github.com/gagliardetto/solana-go" +) + +// BenchmarkEntryPrefetchAssembly runs actual assembler ingestion, component +// decoding, final Merkle identity checks, transaction verification, and final +// publication. Source transactions use the same generated 228/1232-byte or +// captured fixtures as BenchmarkTransactionVerificationFlow. Generation, +// packet parsing and shred-signature authentication happen outside the timer. +// There is no network I/O, loss/recovery, replay, PoH or footer-bankhash execution. +// +// Complete component bursts are scheduled across 200 ms for tip scenarios; +// catchup offers all shreds immediately. This is a workload model, not a +// measured cluster arrival distribution. A 60 KiB signed-transaction budget +// defines components; actual entry overhead and FEC padding are generated. +// Pools persist across iterations; the identical fixture slot is reset between +// samples outside the timer. Use fixed counts (e.g. -benchtime=3x). +func BenchmarkEntryPrefetchAssembly(b *testing.B) { + flowConfigureBackend(b) + for _, source := range flowBenchmarkFixtures(b) { + b.Run(source.name, func(b *testing.B) { + fixture := makeAssemblyFlowFixture(b, source.blk) + for _, workers := range []int{2, 4} { + b.Run(fmt.Sprintf("workers_%d", workers), func(b *testing.B) { + for _, target := range []int{4, 8} { + b.Run(fmt.Sprintf("target_%d", target), func(b *testing.B) { + for _, arrival := range []struct { + name string + span time.Duration + }{{"catchup", 0}, {"tip_200ms", 200 * time.Millisecond}} { + b.Run(arrival.name, func(b *testing.B) { + for _, overlap := range []bool{false, true} { + name := "overlap_off" + if overlap { + name = "overlap_on" + } + b.Run(name, func(b *testing.B) { + runAssemblyFlowBenchmark(b, source.blk, fixture, workers, target, arrival.span, overlap, false) + }) + } + }) + } + }) + } + }) + } + }) + } +} + +type assemblyFlowFixture struct { + components [][]*Shred + authenticatedRoots [][]solana.Hash + blockID solana.Hash + parentID solana.Hash + bankhash solana.Hash + shreds int + holdGap bool +} + +func makeAssemblyFlowFixture(tb testing.TB, source *block.Block) assemblyFlowFixture { + tb.Helper() + fixture := assemblyFlowFixture{parentID: solana.Hash{31}, bankhash: solana.Hash{47}} + var seed [ed25519.SeedSize]byte + seed[0] = 193 // Public deterministic benchmark leader, not validator material. + leader := solana.PrivateKey(ed25519.NewKeyFromSeed(seed[:])) + public := solana.PublicKeyFromBytes(ed25519.PrivateKey(leader).Public().(ed25519.PublicKey)) + generator := ShredGenerator{Slot: 100, ParentSlot: 99, Version: 7, ReferenceTick: 63} + var nextData, nextCode uint32 + var chained solana.Hash + var roots []solana.Hash + appendComponent := func(component BlockComponent, last bool) { + raw, err := MarshalBlockComponent(component) + if err != nil { + tb.Fatal(err) + } + packets, finalRoot, dataEnd, codeEnd, err := generator.MakeShredsFromData(leader, raw, last, chained, nextData, nextCode) + if err != nil { + tb.Fatal(err) + } + chained, nextData, nextCode = finalRoot, dataEnd, codeEnd + var data []*Shred + var authenticatedRoots []solana.Hash + for _, packet := range packets { + shred, err := ParseShred(packet) + if err != nil { + tb.Fatal(err) + } + if shred.Type != ShredTypeData { + continue + } + if err := shred.VerifySignature(public); err != nil { + tb.Fatal(err) + } + root, err := shred.MerkleRoot() + if err != nil { + tb.Fatal(err) + } + if shred.Index == shred.FECSetIndex { + roots = append(roots, root) + } + authenticatedRoots = append(authenticatedRoots, root) + data = append(data, shred) + } + fixture.components = append(fixture.components, data) + fixture.authenticatedRoots = append(fixture.authenticatedRoots, authenticatedRoots) + fixture.shreds += len(data) + } + appendComponent(NewBlockHeader(99, fixture.parentID), false) + for i, txs := range flowComponents(source.Transactions, 60*1024) { + entry := Entry{NumHashes: 1, Txns: make([]solana.Transaction, len(txs))} + binary.LittleEndian.PutUint64(entry.Hash[:], uint64(i+1)) + for j, tx := range txs { + entry.Txns[j] = *tx + } + appendComponent(BlockComponent{EntryBatch: []Entry{entry}}, false) + } + appendComponent(NewBlockFooter(BlockFooter{BankHash: fixture.bankhash}), true) + if nextData > maxDataShredsPerSlot { + tb.Fatalf("fixture requires %d data shreds; assembler limit is %d", nextData, maxDataShredsPerSlot) + } + fixture.blockID = DoubleMerkleBlockID(99, fixture.parentID, roots) + return fixture +} + +func runAssemblyFlowBenchmark(b *testing.B, source *block.Block, fixture assemblyFlowFixture, workers, target int, span time.Duration, overlap, prepareIdentities bool) { + v := newTransactionVerifierWithBatchTarget(workers, 2*workers*target, target, nil) + defer v.closeAndWait() + if err := v.verifyBlock(source); err != nil { + b.Fatal(err) + } + a := NewSlotAssembler() + a.verifyTransactions = v.verifyBlockContext + a.SetKnownAlpenglowBlockID(99, fixture.parentID) + a.SetKnownAlpenglowBlockID(100, fixture.blockID) + if overlap { + prefetch := newEntryPrefetchPool(context.Background(), a, v) + defer prefetch.closeAndWait() + } + var ready, parse, preparation, joins, arrivals, identityPreparation, fullToIdentities []time.Duration + var early uint64 + var cpu float64 + before := sigverify.Stats() + b.ReportAllocs() + b.ResetTimer() + for range b.N { + b.StopTimer() + a.ResetSlot(100) + cpuStarted := flowCPUSeconds(b) + b.StartTimer() + started := time.Now() + var work *slotCompletionWork + for componentIndex, component := range fixture.components { + if span > 0 { + time.Sleep(time.Until(started.Add(flowArrivalOffset(componentIndex, len(fixture.components), span)))) + } + for shredIndex, shred := range component { + if fixture.holdGap && componentIndex == len(fixture.components)*3/4 && shredIndex == 1 { + continue + } + candidate, err := a.addShredFromWithRoot(shred, false, &fixture.authenticatedRoots[componentIndex][shredIndex]) + if err != nil { + b.Fatal(err) + } + if candidate != nil { + if work != nil { + b.Fatal("slot claimed completion more than once") + } + work = candidate + } + } + } + if fixture.holdGap { + ci := len(fixture.components) * 3 / 4 + candidate, err := a.addShredFromWithRoot(fixture.components[ci][1], false, &fixture.authenticatedRoots[ci][1]) + if err != nil || candidate == nil || work != nil { + b.Fatalf("held gap completion failed: %v", err) + } + work = candidate + } + if work == nil { + b.Fatal("all generated data shreds did not complete the slot") + } + processed := a.processCompletion(context.Background(), work) + completed, err := a.finalizeCompletion(work, processed) + if prepareIdentities && err == nil && completed != nil { + start := time.Now() + _, err = completed.PrepareTransactionMessageIdentities() + identityPreparation = append(identityPreparation, time.Since(start)) + fullToIdentities = append(fullToIdentities, time.Since(work.state.fullAt)) + } + b.StopTimer() + cpu += flowCPUSeconds(b) - cpuStarted + if err != nil || completed == nil { + b.Fatalf("completion failed: block=%v error=%v", completed != nil, err) + } + if !completed.TransactionSignaturesVerified() || len(completed.Transactions) != len(source.Transactions) || + !completed.HasAlpenglowBlockID || solana.Hash(completed.AlpenglowBlockID) != fixture.blockID || + !completed.HasExpectedBankhash || completed.ExpectedBankhash != fixture.bankhash { + b.Fatal("completion changed transaction coverage or authenticated block metadata") + } + ready = append(ready, processed.timings.FullToReady) + parse = append(parse, processed.timings.TransactionParse) + preparation = append(preparation, processed.timings.EarlyPreparationWait) + joins = append(joins, processed.timings.TransactionSigverify) + arrivals = append(arrivals, processed.timings.ShredCollection) + early += processed.timings.EarlyVerifiedTransactions + } + after := sigverify.Stats() + b.ReportMetric(cpu*1000/float64(b.N), "cpu-ms/block") + b.ReportMetric(cpu/b.Elapsed().Seconds(), "avg_cpu_cores") + b.ReportMetric(float64(early)/float64(b.N), "early_verified_tx/block") + b.ReportMetric(float64(len(source.Transactions)), "tx/block") + b.ReportMetric(float64(fixture.shreds), "data_shreds/block") + b.ReportMetric(float64(len(fixture.components)), "components/block") + if batches := after.Batches - before.Batches; batches > 0 { + b.ReportMetric(float64(after.Signatures-before.Signatures)/float64(batches), "mean_width") + } + flowReportPercentiles(b, ready, "full_to_ready") + flowReportPercentiles(b, parse, "completion_parse") + flowReportPercentiles(b, preparation, "preparation_wait") + flowReportPercentiles(b, joins, "completion_sigverify") + flowReportPercentiles(b, arrivals, "collection") + if prepareIdentities { + flowReportPercentiles(b, identityPreparation, "identity_admission") + flowReportPercentiles(b, fullToIdentities, "full_to_identities") + } + if after.InternalFaultFallbacks != before.InternalFaultFallbacks { + b.Fatal("signature verifier used an internal fault fallback") + } + var signatures uint64 + for _, tx := range source.Transactions { + signatures += uint64(len(tx.Signatures)) + } + if got, want := after.Signatures-before.Signatures, signatures*uint64(b.N); got != want { + b.Fatalf("verified %d signatures; want %d, exactly once per retained transaction", got, want) + } +} + +// BenchmarkEntryMessageIdentityArrival includes the first admission-time message +// identity lookup after assembly. Compare unchanged baseline and candidate with +// identical fixtures; full_to_identities includes any moved completion work. +// This does not include replay's whole-block duplicate map or transaction loop. +func BenchmarkEntryMessageIdentityArrival(b *testing.B) { + flowConfigureBackend(b) + for _, source := range flowBenchmarkFixtures(b) { + b.Run(source.name, func(b *testing.B) { + fixture := makeAssemblyFlowFixture(b, source.blk) + for _, arrival := range []struct { + name string + span time.Duration + }{{"catchup", 0}, {"tip_200ms", 200 * time.Millisecond}} { + b.Run(arrival.name, func(b *testing.B) { + runAssemblyFlowBenchmark(b, source.blk, fixture, 2, 8, arrival.span, true, true) + }) + } + }) + } +} + +// Holds one data shred in a batch three quarters through the block until the +// footer has arrived. Other complete batches continue arriving over 200 ms. +// This isolates discovery behind a gap; it is not a measured network replay. +func BenchmarkEntryPrefetchGapArrival(b *testing.B) { + flowConfigureBackend(b) + for _, source := range flowBenchmarkFixtures(b) { + b.Run(source.name, func(b *testing.B) { + fixture := makeAssemblyFlowFixture(b, source.blk) + for _, gap := range []bool{false, true} { + b.Run(fmt.Sprintf("gap_%t", gap), func(b *testing.B) { + fixture.holdGap = gap + runAssemblyFlowBenchmark(b, source.blk, fixture, 2, 8, 200*time.Millisecond, true, true) + }) + } + }) + } +} diff --git a/pkg/turbine/entry_prefetch_bounds_test.go b/pkg/turbine/entry_prefetch_bounds_test.go new file mode 100644 index 000000000..f7c9aef4a --- /dev/null +++ b/pkg/turbine/entry_prefetch_bounds_test.go @@ -0,0 +1,135 @@ +package turbine + +import ( + "context" + "fmt" + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +func TestEntryPrefetchByteBoundsFallBackToCompleteVerification(t *testing.T) { + for _, mode := range []string{"budget_full", "oversized_component"} { + t.Run(mode, func(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + if mode == "budget_full" { + // Model other generations owning the entire encoded-byte budget. + a.mu.Lock() + p.bytes = entryPrefetchBytes + a.mu.Unlock() + defer func() { a.mu.Lock(); p.bytes -= entryPrefetchBytes; a.mu.Unlock() }() + } + raw := prefetchTestPayload(t, verifierSignedTransactions(t, 3)) + if mode == "oversized_component" { + // Ordinary entry components permit trailing FEC padding. The + // complete decoder must accept this even though prefetch skips it. + raw = append(raw, make([]byte, entryPrefetchBatchBytes+1-len(raw))...) + } + const slot = 400 + batches := prefetchTestShreds(t, slot, raw, buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + f := a.slots[slot].prefetch + return f != nil && !f.queued && len(f.batches) == 0 + }, 3*time.Second, time.Millisecond) + if mode == "oversized_component" { + a.mu.Lock() + state := a.slots[slot] + reason, released := state.streamCancelReason, state.prefetch.released + a.mu.Unlock() + require.Equal(t, "prefetch_batch_too_large", reason) + require.True(t, released) + require.Equal(t, StreamGone, a.StreamStatusOf(StreamGeneration{slot: slot, state: state})) + } + blk := feedPrefetchShreds(t, a, batches[1]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 3) + require.True(t, blk.TransactionSignaturesVerified()) + timings, ok := blk.TurbineIngressTimings() + require.True(t, ok) + require.Zero(t, timings.EarlyVerifiedTransactions) + }) + } +} + +// A closed range needs no additional shred to become eligible after another +// generation releases its reservation. Cancellation must not revive stale work. +func TestEntryPrefetchRetriesByteBudgetOnRelease(t *testing.T) { + for _, cancelWaiting := range []bool{false, true} { + t.Run(fmt.Sprint(cancelWaiting), func(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + holderCtx, cancel := context.WithCancel(context.Background()) + holder := &slotState{slot: 399, prefetch: &slotEntryPrefetch{pool: p, ctx: holderCtx, cancel: cancel, bytes: entryPrefetchBytes}} + a.mu.Lock() + a.slots[399] = holder + p.slots = 1 + p.bytes = entryPrefetchBytes + p.active = map[*slotState]struct{}{holder: {}} + a.mu.Unlock() + batches := prefetchTestShreds(t, 400, prefetchTestPayload(t, verifierSignedTransactions(t, 3)), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + f := a.slots[400].prefetch + return f != nil && f.budgetBlocked && !f.queued + }, 3*time.Second, time.Millisecond) + if cancelWaiting { + a.ResetSlot(400) + } + a.ResetSlot(399) + if cancelWaiting { + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + return p.slots == 0 && p.bytes == 0 && len(p.active) == 0 + }, 3*time.Second, time.Millisecond) + } else { + batch := waitPrefetchedBatch(t, a, 400, 0) + _, err := batch.verification.wait() + require.NoError(t, err) + require.Len(t, entryBatchTransactions(batch.entries), 3) + } + }) + } +} + +func TestEntryPrefetchOversizedRangeAfterVerifiedPrefix(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + raw := prefetchTestPayload(t, verifierSignedTransactions(t, 3)) + large := prefetchTestPayload(t, verifierSignedTransactions(t, 3)) + large = append(large, make([]byte, entryPrefetchBatchBytes+1-len(large))...) + batches := prefetchTestShreds(t, 401, raw, large, buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + prefix := waitPrefetchedBatch(t, a, 401, 0) + _, err := prefix.verification.wait() + require.NoError(t, err) + require.Nil(t, feedPrefetchShreds(t, a, batches[1])) + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + return a.slots[401].prefetch.released + }, 3*time.Second, time.Millisecond) + blk := feedPrefetchShreds(t, a, batches[2]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 6) + require.True(t, blk.TransactionSignaturesVerified()) + timings, ok := blk.TurbineIngressTimings() + require.True(t, ok) + require.Zero(t, timings.EarlyVerifiedTransactions, "cancelled prefix cache is not reused") +} diff --git a/pkg/turbine/entry_prefetch_test.go b/pkg/turbine/entry_prefetch_test.go new file mode 100644 index 000000000..9dbbd9ada --- /dev/null +++ b/pkg/turbine/entry_prefetch_test.go @@ -0,0 +1,691 @@ +package turbine + +import ( + "context" + "errors" + "fmt" + "runtime" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func prefetchTestPayload(t *testing.T, txs []*solana.Transaction) []byte { + t.Helper() + entry := Entry{NumHashes: 1, Hash: solana.Hash{7}, Txns: make([]solana.Transaction, len(txs))} + for i, tx := range txs { + entry.Txns[i] = *tx + } + raw, err := marshalEntryBatch([]Entry{entry}) + require.NoError(t, err) + return raw +} + +func prefetchTestShreds(t *testing.T, slot uint64, payloads ...[]byte) [][]*Shred { + t.Helper() + gen := ShredGenerator{Slot: slot, ParentSlot: slot - 1, Version: 1} + var nextData, nextCode uint32 + var root solana.Hash + batches := make([][]*Shred, len(payloads)) + for i, raw := range payloads { + packets, nextRoot, data, code, err := gen.MakeShredsFromData(testShredLeader(t), raw, i == len(payloads)-1, root, nextData, nextCode) + require.NoError(t, err) + root, nextData, nextCode = nextRoot, data, code + for _, packet := range packets { + shred, err := ParseShred(packet) + require.NoError(t, err) + if shred.Type == ShredTypeData { + batches[i] = append(batches[i], shred) + } + } + } + return batches +} + +func feedPrefetchShreds(t *testing.T, a *SlotAssembler, shreds []*Shred) *block.Block { + t.Helper() + var result *block.Block + for _, shred := range shreds { + blk, err := a.AddShred(shred) + require.NoError(t, err) + if blk != nil { + require.Nil(t, result) + result = blk + } + } + return result +} + +func waitPrefetchedBatch(t *testing.T, a *SlotAssembler, slot uint64, start uint32) *prefetchedShredBatch { + t.Helper() + var batch *prefetchedShredBatch + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + if s := a.slots[slot]; s != nil && s.prefetch != nil { + batch = s.prefetch.batches[start] + } + return batch != nil + }, 3*time.Second, time.Millisecond) + waitSignal(t, batch.ready, "prefetched component preparation") + require.NoError(t, batch.err) + require.NoError(t, batch.submitErr) + return batch +} + +func TestEntryPrefetchVerifiesBeforeLastShredAndReusesResults(t *testing.T) { + var calls atomic.Int32 + v := newTransactionVerifier(2, 16, func(tx *solana.Transaction) error { + calls.Add(1) + return txverify.VerifyTransaction(tx) + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 100 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), + prefetchTestPayload(t, verifierSignedTransactions(t, 4))) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, slot, 0) + require.NotNil(t, cached.verification) + _, err := cached.verification.wait() + require.NoError(t, err) + require.Equal(t, int32(3), calls.Load()) + for range 5 { + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + } + blk := feedPrefetchShreds(t, a, batches[1]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 7) + require.True(t, blk.TransactionSignaturesVerified()) + require.Equal(t, int32(7), calls.Load(), "cached transactions must not be verified twice") + timings, ok := blk.TurbineIngressTimings() + require.True(t, ok) + require.Equal(t, uint64(3), timings.EarlyVerifiedTransactions) + require.LessOrEqual(t, cached.verification.finishedAt.UnixNano(), blk.ShredFullNanos) +} + +func TestEntryPrefetchWaitsForGapAcrossMultipleFECSets(t *testing.T) { + var calls atomic.Int32 + v := newTransactionVerifier(2, 16, func(*solana.Transaction) error { calls.Add(1); return nil }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 104 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 300)), buildAlpenglowEndingTick(t)) + require.Greater(t, len(batches[0]), dataShredsPerFECBlock) + // Arrival of DATA_COMPLETE and later FEC sets cannot bypass the hole. + for i := len(batches[0]) - 1; i >= 0; i-- { + if i != 1 { + require.Nil(t, feedPrefetchShreds(t, a, batches[0][i:i+1])) + } + } + a.mu.Lock() + require.Empty(t, a.slots[slot].completeBatches) + require.Nil(t, a.slots[slot].prefetch) + a.mu.Unlock() + require.Zero(t, calls.Load()) + require.Nil(t, feedPrefetchShreds(t, a, batches[0][1:2])) + cached := waitPrefetchedBatch(t, a, slot, 0) + _, err := cached.verification.wait() + require.NoError(t, err) + require.Equal(t, int32(300), calls.Load()) + blk := feedPrefetchShreds(t, a, batches[1]) + require.NotNil(t, blk) + require.Equal(t, int32(300), calls.Load()) +} + +func TestEntryPrefetchResetKeepsOldReservationsUntilReadersJoin(t *testing.T) { + txs := verifierSignedTransactions(t, 2) + txs[0].Signatures[0][9] ^= 0x40 + oldSignature := txs[0].Signatures[0] + started := make(chan struct{}) + release := make(chan struct{}) + var releaseOnce sync.Once + v := newTransactionVerifier(1, 8, func(tx *solana.Transaction) error { + if tx.Signatures[0] == oldSignature { + close(started) + <-release + } + return txverify.VerifyTransaction(tx) + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + const slot = 108 + old := prefetchTestShreds(t, slot, prefetchTestPayload(t, txs[:1]), buildAlpenglowEndingTick(t)) + fresh := prefetchTestShreds(t, slot, prefetchTestPayload(t, txs[1:]), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, old[0])) + waitSignal(t, started, "old generation verifier") + oldBatch := waitPrefetchedBatch(t, a, slot, 0) + a.ResetSlot(slot) + a.mu.Lock() + require.Equal(t, 1, p.slots) + require.Positive(t, p.bytes) + a.mu.Unlock() + require.Nil(t, feedPrefetchShreds(t, a, fresh[0])) + a.mu.Lock() + require.Equal(t, 2, p.slots) + a.mu.Unlock() + // With one verifier worker only one request may prefetch. The fresh + // generation keeps its pool reservation while admission waits for the old + // reader to join; it must not release or reuse the old generation's bytes. + releaseOnce.Do(func() { close(release) }) + _, err := oldBatch.verification.wait() + require.ErrorIs(t, err, context.Canceled) + newBatch := waitPrefetchedBatch(t, a, slot, 0) + require.NotSame(t, oldBatch, newBatch) + _, err = newBatch.verification.wait() + require.NoError(t, err) + blk := feedPrefetchShreds(t, a, fresh[1]) + require.NotNil(t, blk) + require.Equal(t, txs[1].Signatures[0], blk.Transactions[0].Signatures[0]) + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + return p.slots == 0 && p.bytes == 0 + }, 3*time.Second, time.Millisecond) +} + +func TestEntryPrefetchSaturationDoesNotBlockShredAdmissionAndShutdownJoins(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + var first, releaseOnce sync.Once + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { + first.Do(func() { close(started) }) + <-release + return nil + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + payload := prefetchTestPayload(t, verifierSignedTransactions(t, 1)) + for i := 0; i < entryPrefetchSlots+1; i++ { + batches := prefetchTestShreds(t, uint64(200+i), payload, buildAlpenglowEndingTick(t)) + admitted := make(chan struct{}) + go func() { + defer close(admitted) + for _, sh := range batches[0] { + _, err := a.addShredFrom(sh, false) + if err != nil { + t.Errorf("admit shred: %v", err) + } + } + }() + waitSignal(t, admitted, "nonblocking ingress while prefetch saturated") + } + waitSignal(t, started, "blocked verifier") + a.mu.Lock() + require.Equal(t, entryPrefetchSlots, p.slots) + require.LessOrEqual(t, p.bytes, entryPrefetchBytes) + require.Nil(t, a.slots[200+entryPrefetchSlots].prefetch) + a.mu.Unlock() + done := make(chan struct{}) + go func() { p.closeAndWait(); close(done) }() + select { + case <-done: + t.Fatal("early pool released buffers before signature worker joined") + case <-time.After(20 * time.Millisecond): + } + releaseOnce.Do(func() { close(release) }) + waitSignal(t, done, "saturated pool shutdown") + a.mu.Lock() + require.Zero(t, p.slots) + require.Zero(t, p.bytes) + require.Nil(t, a.entryPrefetch) + a.mu.Unlock() +} + +func TestEntryPrefetchInvalidRetainedTransactionFailsClosed(t *testing.T) { + txs := verifierSignedTransactions(t, 3) + txs[1].Signatures[0][3] ^= 0x80 + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + batches := prefetchTestShreds(t, 300, prefetchTestPayload(t, txs), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, 300, 0) + index, err := cached.verification.wait() + require.Equal(t, 1, index) + require.ErrorContains(t, err, "invalid signature") + var finalErr error + for _, sh := range batches[1] { + blk, err := a.AddShred(sh) + require.Nil(t, blk) + if err != nil { + finalErr = err + } + } + require.ErrorContains(t, finalErr, "transaction 1") + require.False(t, a.SlotCompleted(300)) + p.cleanup.Wait() + a.mu.Lock() + g := StreamGeneration{slot: 300, state: a.slots[300]} + slots, bytes := p.slots, p.bytes + a.mu.Unlock() + require.Equal(t, StreamGone, a.StreamStatusOf(g)) + require.Zero(t, slots) + require.Zero(t, bytes) +} + +func TestEntryPrefetchUpdateParentDiscardsInvalidOptimisticPrefix(t *testing.T) { + txs := verifierSignedTransactions(t, 2) + txs[0].Signatures[0][3] ^= 0x80 + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 304 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, txs[:1]), + testAlpenglowParentMarkerBytes(blockMarkerVariantUpdateParent, slot-2, solana.Hash{12}), + prefetchTestPayload(t, txs[1:])) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, slot, 0) + _, err := cached.verification.wait() + require.ErrorContains(t, err, "invalid signature") + require.Nil(t, feedPrefetchShreds(t, a, batches[1])) + blk := feedPrefetchShreds(t, a, batches[2]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 1) + require.Equal(t, txs[1].Signatures[0], blk.Transactions[0].Signatures[0]) + require.Equal(t, uint64(slot-2), blk.SourceParentSlot) + require.True(t, blk.TransactionSignaturesVerified()) +} + +func TestEntryPrefetchCanceledCompletionCanRetrySameGeneration(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + var first, releaseOnce sync.Once + v := newTransactionVerifier(1, 8, func(tx *solana.Transaction) error { + first.Do(func() { close(started); <-release }) + return txverify.VerifyTransaction(tx) + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + const slot = 308 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + waitSignal(t, started, "early verification") + cached := waitPrefetchedBatch(t, a, slot, 0) + var work *slotCompletionWork + for _, sh := range batches[1] { + var err error + work, err = a.addShredFrom(sh, false) + require.NoError(t, err) + } + require.NotNil(t, work) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + canceled := make(chan struct{}) + originalCancel := cached.verification.cancel + cached.verification.cancel = func() { close(canceled); originalCancel() } + done := make(chan processedSlotCompletion, 1) + go func() { done <- a.processCompletion(ctx, work) }() + // The completion has no expensive decode left and blocks joining this + // one already-prepared future; cancel while it owns those transactions. + require.Eventually(t, func() bool { + stack := make([]byte, 2<<20) + for _, goroutine := range strings.Split(string(stack[:runtime.Stack(stack, true)]), "\n\n") { + if strings.Contains(goroutine, "(*SlotAssembler).processCompletion") && strings.Contains(goroutine, "(*transactionVerification).waitContext") { + return true + } + } + return false + }, 3*time.Second, time.Millisecond, "completion must enter the owning verification join") + cancel() + waitSignal(t, canceled, "completion canceled its signature request") + releaseOnce.Do(func() { close(release) }) + var processed processedSlotCompletion + select { + case processed = <-done: + case <-time.After(3 * time.Second): + t.Fatal("canceled completion failed to join") + } + require.True(t, processed.canceled) + _, err := a.finalizeCompletion(work, processed) + require.NoError(t, err) + _, err = cached.verification.wait() + require.True(t, errors.Is(err, context.Canceled)) + a.mu.Lock() + retry := a.claimCompletionLocked(a.slots[slot], false) + a.mu.Unlock() + require.NotNil(t, retry) + processed = a.processCompletion(context.Background(), retry) + require.NoError(t, processed.err) + blk, err := a.finalizeCompletion(retry, processed) + require.NoError(t, err) + require.NotNil(t, blk) + require.True(t, blk.TransactionSignaturesVerified()) +} + +// A complete later batch must verify while an earlier batch still has a gap. +// Its preceding DATA_COMPLETE shred remains necessary to establish its start. +func TestEntryPrefetchBypassesEarlierGap(t *testing.T) { + for _, lateBoundary := range []bool{false, true} { + t.Run(fmt.Sprint("lateBoundary=", lateBoundary), func(t *testing.T) { + var calls atomic.Int32 + v := newTransactionVerifier(2, 8, func(tx *solana.Transaction) error { calls.Add(1); return txverify.VerifyTransaction(tx) }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 909 + txs := verifierSignedTransactions(t, 60) + batches := prefetchTestShreds(t, slot, prefetchTestPayload(t, txs[:30]), prefetchTestPayload(t, txs[30:]), buildAlpenglowEndingTick(t)) + require.Greater(t, len(batches[0]), 2) + end := len(batches[0]) - 1 + for i, sh := range batches[0] { + if i == 1 || (lateBoundary && i == end) { + continue + } + require.Nil(t, feedPrefetchShreds(t, a, []*Shred{sh})) + } + require.Nil(t, feedPrefetchShreds(t, a, batches[1])) + if lateBoundary { + a.mu.Lock() + empty := len(a.slots[slot].completeBatches) == 0 + a.mu.Unlock() + require.True(t, empty, "unknown preceding boundary must prevent speculation") + require.Nil(t, feedPrefetchShreds(t, a, batches[0][end:])) + } + later := waitPrefetchedBatch(t, a, slot, batches[1][0].Index) + _, err := later.verification.wait() + require.NoError(t, err) + require.Equal(t, int32(30), calls.Load()) + require.Nil(t, feedPrefetchShreds(t, a, batches[0][1:2])) + earlier := waitPrefetchedBatch(t, a, slot, 0) + _, err = earlier.verification.wait() + require.NoError(t, err) + blk := feedPrefetchShreds(t, a, batches[2]) + require.NotNil(t, blk) + require.True(t, blk.TransactionSignaturesVerified()) + require.Len(t, blk.Transactions, 60) + require.Equal(t, int32(60), calls.Load(), "every signature verified exactly once") + for i, tx := range txs { + require.Equal(t, tx.Signatures[0], blk.Transactions[i].Signatures[0]) + } + }) + } +} + +func TestEntryPrefetchDiscoversRecoveredBoundary(t *testing.T) { + v := newTransactionVerifier(2, 8, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 910 + txs := verifierSignedTransactions(t, 60) + gen := ShredGenerator{Slot: slot, ParentSlot: slot - 1, Version: 1} + raw := prefetchTestPayload(t, txs[:30]) + packets, root, nextData, nextCode, err := gen.MakeShredsFromData(testShredLeader(t), raw, false, solana.Hash{}, 0, 0) + require.NoError(t, err) + var code []*Shred + for _, packet := range packets { + sh, err := ParseShred(packet) + require.NoError(t, err) + if sh.Type == ShredTypeCode { + code = append(code, sh) + continue + } + if !sh.DataComplete() { + require.Nil(t, feedPrefetchShreds(t, a, []*Shred{sh})) + } + } + packets, _, _, _, err = gen.MakeShredsFromData(testShredLeader(t), prefetchTestPayload(t, txs[30:]), false, root, nextData, nextCode) + require.NoError(t, err) + for _, packet := range packets { + sh, err := ParseShred(packet) + require.NoError(t, err) + if sh.Type == ShredTypeData { + require.Nil(t, feedPrefetchShreds(t, a, []*Shred{sh})) + } + } + a.mu.Lock() + empty := len(a.slots[slot].completeBatches) == 0 + a.mu.Unlock() + require.True(t, empty) + require.NotEmpty(t, code) + for _, sh := range code { + require.Nil(t, feedPrefetchShreds(t, a, []*Shred{sh})) + } + later := waitPrefetchedBatch(t, a, slot, nextData) + _, err = later.verification.wait() + require.NoError(t, err) + first := waitPrefetchedBatch(t, a, slot, 0) + _, err = first.verification.wait() + require.NoError(t, err) +} + +func TestEntryPrefetchIndexDisabledAndLateInstall(t *testing.T) { + a := NewSlotAssembler() + batches := prefetchTestShreds(t, 911, prefetchTestPayload(t, verifierSignedTransactions(t, 3)), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + a.mu.Lock() + absent := a.slots[911].batchIndex == nil + a.mu.Unlock() + require.True(t, absent) + v := newTransactionVerifier(2, 8, nil) + defer v.closeAndWait() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + // Seeding also works on a coding-only admission with no new data recovery. + a.mu.Lock() + a.prefetchEntriesLocked(a.slots[911]) + a.mu.Unlock() + cached := waitPrefetchedBatch(t, a, 911, 0) + _, err := cached.verification.wait() + require.NoError(t, err) +} + +func TestEntryPrefetchResetRetainsQueuedReservations(t *testing.T) { + a := NewSlotAssembler() + ctx, cancel := context.WithCancel(context.Background()) + // Hold workers until after reset so all tokens remain in the channel. + p := &entryPrefetchPool{a: a, ctx: ctx, cancel: cancel, jobs: make(chan *slotState, entryPrefetchSlots)} + a.entryPrefetch = p + startWorker := sync.OnceFunc(func() { + p.workers.Add(1) + go p.run() + }) + defer func() { + startWorker() + p.closeAndWait() + }() + for i := 0; i < entryPrefetchSlots; i++ { + s := &slotState{slot: uint64(i), completeBatches: []shredBatchRange{{0, 0}}} + a.mu.Lock() + a.slots[s.slot] = s + a.prefetchEntriesLocked(s) + a.releasePrefetchLocked(s) + a.mu.Unlock() + } + require.Equal(t, entryPrefetchSlots, len(p.jobs)) + // Cleanup must not admit another generation while stale queue tokens live. + require.Never(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + return p.slots != entryPrefetchSlots + }, 50*time.Millisecond, time.Millisecond) + fresh := &slotState{slot: 100, completeBatches: []shredBatchRange{{0, 0}}} + a.mu.Lock() + a.prefetchEntriesLocked(fresh) + reserved := fresh.prefetch != nil + a.mu.Unlock() + // Start the worker before assertions so test failures cannot strand cleanup. + startWorker() + require.False(t, reserved) + p.cleanup.Wait() + a.mu.Lock() + slots := p.slots + a.mu.Unlock() + require.Zero(t, slots) +} + +// Failed full blocks remain available for diagnostics, but must not consume +// the prefetch budget or remain usable streaming generations until retention. +func TestEntryPrefetchFailedCompletionReleasesCapacity(t *testing.T) { + for _, dropEvents := range []bool{false, true} { + t.Run(fmt.Sprintf("drop_events=%v", dropEvents), func(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + capacity := 16 + if dropEvents { + capacity = 0 + } + events := make(chan StreamEvent, capacity) + a.SubscribeStream(events) + payload := prefetchTestPayload(t, verifierSignedTransactions(t, 1)) + for i := 0; i <= entryPrefetchSlots; i++ { + slot := uint64(900 + i) + // Zero entry count plus trailing bytes is an invalid component. + batches := prefetchTestShreds(t, slot, payload, make([]byte, 16)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + batch := waitPrefetchedBatch(t, a, slot, 0) + _, err := batch.verification.wait() + require.NoError(t, err) + a.mu.Lock() + s := a.slots[slot] + g := StreamGeneration{slot: slot, state: s} + a.mu.Unlock() + var failure error + for _, sh := range batches[1] { + blk, err := a.AddShred(sh) + require.Nil(t, blk) + if err != nil { + failure = err + } + } + require.Error(t, failure) + count, last := a.SlotAssemblyErrors(slot) + require.Positive(t, count) + require.Equal(t, failure.Error(), last) + require.False(t, a.SlotCompleted(slot)) + require.Equal(t, StreamGone, a.StreamStatusOf(g)) + require.Empty(t, a.PendingStreamBatches(g, 0)) + if !dropEvents { + event := nextStreamEvent(t, events, StreamCancelled) + require.Equal(t, g, event.Generation) + require.Equal(t, "completion_failed", event.Reason) + } + p.cleanup.Wait() + a.mu.Lock() + retained, slots, bytes := a.slots[slot], p.slots, p.bytes + // Repeated release/admission cannot double-refund or resurrect it. + a.releasePrefetchLocked(s) + a.prefetchEntriesLocked(s) + afterSlots := p.slots + a.mu.Unlock() + require.Same(t, s, retained, "preserve poisoned-slot diagnostics") + require.Zero(t, slots) + require.Zero(t, bytes) + require.Zero(t, afterSlots) + } + good := prefetchTestShreds(t, 920, payload, buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, good[0])) + waitPrefetchedBatch(t, a, 920, 0) + blk := feedPrefetchShreds(t, a, good[1]) + require.NotNil(t, blk) + require.True(t, blk.TransactionSignaturesVerified()) + }) + } +} + +func TestEntryPrefetchFailedCompletionJoinsReaders(t *testing.T) { + started, release := make(chan struct{}), make(chan struct{}) + var once sync.Once + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { + close(started) + <-release + return nil + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + defer once.Do(func() { close(release) }) + batches := prefetchTestShreds(t, 930, prefetchTestPayload(t, verifierSignedTransactions(t, 1)), make([]byte, 16)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + waitSignal(t, started, "prefetch verifier") + waitPrefetchedBatch(t, a, 930, 0) + var failure error + for _, sh := range batches[1] { + blk, err := a.AddShred(sh) + require.Nil(t, blk) + if err != nil { + failure = err + } + } + require.Error(t, failure) + a.mu.Lock() + s := a.slots[930] + released, ctxErr, slots, bytes := s.prefetch.released, s.prefetch.ctx.Err(), p.slots, p.bytes + a.mu.Unlock() + require.True(t, released) + require.ErrorIs(t, ctxErr, context.Canceled) + require.Equal(t, 1, slots, "reader still owns the reservation") + require.Positive(t, bytes) + require.Equal(t, StreamGone, a.StreamStatusOf(StreamGeneration{slot: 930, state: s})) + once.Do(func() { close(release) }) + p.cleanup.Wait() + a.mu.Lock() + slots, bytes = p.slots, p.bytes + a.mu.Unlock() + require.Zero(t, slots) + require.Zero(t, bytes) +} + +// A failed generation that never received a reservation must not acquire one +// later when capacity becomes available (or prefetch is attached). +func TestEntryPrefetchFailedCompletionWithoutReservation(t *testing.T) { + a := NewSlotAssembler() + batches := prefetchTestShreds(t, 940, prefetchTestPayload(t, verifierSignedTransactions(t, 1)), make([]byte, 16)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + var failure error + for _, sh := range batches[1] { + blk, err := a.AddShred(sh) + require.Nil(t, blk) + if err != nil { + failure = err + } + } + require.Error(t, failure) + v := newTransactionVerifier(1, 8, nil) + defer v.closeAndWait() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + a.mu.Lock() + s := a.slots[940] + a.prefetchEntriesLocked(s) + reserved, slots := s.prefetch, p.slots + a.mu.Unlock() + require.Nil(t, reserved) + require.Zero(t, slots) + require.Equal(t, StreamGone, a.StreamStatusOf(StreamGeneration{slot: 940, state: s})) +} diff --git a/pkg/turbine/fec_root_cache.go b/pkg/turbine/fec_root_cache.go new file mode 100644 index 000000000..cb4395b01 --- /dev/null +++ b/pkg/turbine/fec_root_cache.go @@ -0,0 +1,56 @@ +package turbine + +import ( + "bytes" + + "github.com/gagliardetto/solana-go" +) + +// One snapshot per FEC generation binds an authenticated root to its source. +// It is written only during assembly, before the state is frozen. Completion +// never trusts pointer identity alone or changes the deterministic root choice. +type authenticatedFECRoot struct { + source *Shred + input shredRootInput + payload []byte + root solana.Hash +} + +// These are every parsed field read by MerkleRoot. The exact payload comparison +// covers headers, data, proof, and trailers, even for noncanonical test input. +type shredRootInput struct { + variant byte + kind ShredType + index, fecSetIndex uint32 + dataCount, position uint16 +} + +func rootInput(s *Shred) shredRootInput { + return shredRootInput{s.Variant, s.Type, s.Index, s.FECSetIndex, s.NumDataShreds, s.Position} +} + +func hasMerkleRootProof(s *Shred) bool { + return s != nil && isMerkleVariant(s.Variant) && (s.Type == ShredTypeData || s.Type == ShredTypeCode) +} + +func (c *authenticatedFECRoot) matches(s *Shred) bool { + return s == c.source && rootInput(s) == c.input && bytes.Equal(s.Payload, c.payload) +} + +func (f *fecState) rememberAuthenticatedRoot(s *Shred, root solana.Hash) { + if s.Recovered || !isMerkleVariant(s.Variant) { + return + } + if cached := f.rootCache; cached != nil { + // Data proofs precede coding proofs, and the lowest index wins. An + // unauthenticated earlier arrival can still force fallback at completion. + old := cached.input + if old.kind == ShredTypeData && (s.Type != ShredTypeData || s.Index >= old.index) { + return + } + if old.kind == ShredTypeCode && s.Type == ShredTypeCode && s.Position >= old.position { + return + } + } + f.rootCache = &authenticatedFECRoot{source: s, input: rootInput(s), payload: bytes.Clone(s.Payload), root: root} +} diff --git a/pkg/turbine/generate.go b/pkg/turbine/generate.go index 14626e214..f8a4e6343 100644 --- a/pkg/turbine/generate.go +++ b/pkg/turbine/generate.go @@ -4,6 +4,7 @@ import ( "crypto/ed25519" "encoding/binary" "fmt" + "sync" "github.com/gagliardetto/solana-go" "github.com/klauspost/reedsolomon" @@ -14,6 +15,21 @@ const ( proofEntriesFor32x32 = 6 ) +var erasureEncoderPool sync.Pool + +func acquireErasureEncoder() (reedsolomon.Encoder, error) { + if encoder := erasureEncoderPool.Get(); encoder != nil { + return encoder.(reedsolomon.Encoder), nil + } + return reedsolomon.New(dataShredsPerFECBlock, codingShredsPerFECBlock) +} + +func releaseErasureEncoder(encoder reedsolomon.Encoder) { + if encoder != nil { + erasureEncoderPool.Put(encoder) + } +} + // ShredGenerator builds merkle FEC shreds from a serialized byte buffer. type ShredGenerator struct { Slot uint64 @@ -22,6 +38,14 @@ type ShredGenerator struct { ReferenceTick uint8 } +// shredPackets retains the roots already computed during generation, in FEC +// order, so broadcast can commit to them without parsing the packets again. +type shredPackets struct { + packets [][]byte + fecSetRoots []solana.Hash + chainedMerkleRoot solana.Hash +} + func dataCapacity(proofSize uint8, resigned bool) int { capacity := dataPayloadSize - dataHeaderSize - merkleRootSize - int(proofSize)*merkleProofEntrySize if resigned { @@ -66,9 +90,31 @@ func (g *ShredGenerator) MakeShredsFromData( nextShredIndex uint32, nextCodeIndex uint32, ) ([][]byte, solana.Hash, uint32, uint32, error) { + batch, nextData, nextCode, err := g.makeShredsFromData( + leader, data, isLastInSlot, chainedMerkleRoot, nextShredIndex, nextCodeIndex, + ) + return batch.packets, batch.chainedMerkleRoot, nextData, nextCode, err +} + +func (g *ShredGenerator) makeShredsFromData( + leader solana.PrivateKey, + data []byte, + isLastInSlot bool, + chainedMerkleRoot solana.Hash, + nextShredIndex uint32, + nextCodeIndex uint32, +) (shredPackets, uint32, uint32, error) { if g.Slot < g.ParentSlot || g.Slot-g.ParentSlot > uint64(^uint16(0)) { - return nil, solana.Hash{}, nextShredIndex, nextCodeIndex, fmt.Errorf("invalid parent slot %d for slot %d", g.ParentSlot, g.Slot) + return shredPackets{}, nextShredIndex, nextCodeIndex, fmt.Errorf("invalid parent slot %d for slot %d", g.ParentSlot, g.Slot) + } + // The 32+32 coding matrix is invariant across every FEC set in this + // operation. Building it requires a Vandermonde inversion, so retain the + // encoder for the whole payload rather than reconstructing it per set. + encoder, err := acquireErasureEncoder() + if err != nil { + return shredPackets{}, nextShredIndex, nextCodeIndex, err } + defer releaseErasureEncoder(encoder) proofSize := uint8(proofEntriesFor32x32) unsignedCap := dataCapacity(proofSize, false) signedCap := dataCapacity(proofSize, true) @@ -92,49 +138,62 @@ func (g *ShredGenerator) MakeShredsFromData( } var packets [][]byte + var fecSetRoots []solana.Hash dataIndex := nextShredIndex codeIndex := nextCodeIndex chainedRoot := chainedMerkleRoot + // DATA_COMPLETE ends the serialized component, which may span FEC sets. for len(unsignedData) >= unsignedBatch { batch := unsignedData[:unsignedBatch] unsignedData = unsignedData[unsignedBatch:] - batchPackets, root, err := g.makeFECBatch(leader, batch, unsignedCap, proofSize, false, parentOffset, flags, false, chainedRoot, dataIndex, codeIndex) + // DATA_COMPLETE marks the end of the serialized component, not the end + // of every FEC set. A full unsigned batch is complete only when no + // unsigned remainder or signed-last batch follows it. + dataComplete := len(unsignedData) == 0 && len(signedData) == 0 + batchPackets, root, err := g.makeFECBatch(encoder, leader, batch, unsignedCap, proofSize, false, parentOffset, flags, dataComplete, false, chainedRoot, dataIndex, codeIndex) if err != nil { - return nil, solana.Hash{}, dataIndex, codeIndex, err + return shredPackets{}, dataIndex, codeIndex, err } packets = append(packets, batchPackets...) + fecSetRoots = append(fecSetRoots, root) chainedRoot = root dataIndex += dataShredsPerFECBlock codeIndex += codingShredsPerFECBlock } if len(unsignedData) > 0 || (len(packets) == 0 && !isLastInSlot) { - batchPackets, root, err := g.makeFECBatch(leader, unsignedData, unsignedCap, proofSize, false, parentOffset, flags, false, chainedRoot, dataIndex, codeIndex) + dataComplete := len(signedData) == 0 + batchPackets, root, err := g.makeFECBatch(encoder, leader, unsignedData, unsignedCap, proofSize, false, parentOffset, flags, dataComplete, false, chainedRoot, dataIndex, codeIndex) if err != nil { - return nil, solana.Hash{}, dataIndex, codeIndex, err + return shredPackets{}, dataIndex, codeIndex, err } packets = append(packets, batchPackets...) + fecSetRoots = append(fecSetRoots, root) chainedRoot = root dataIndex += dataShredsPerFECBlock codeIndex += codingShredsPerFECBlock } if len(signedData) > 0 || (len(packets) == 0 && isLastInSlot) { - batchPackets, root, err := g.makeFECBatch(leader, signedData, signedCap, proofSize, true, parentOffset, flags, isLastInSlot, chainedRoot, dataIndex, codeIndex) + batchPackets, root, err := g.makeFECBatch(encoder, leader, signedData, signedCap, proofSize, true, parentOffset, flags, true, isLastInSlot, chainedRoot, dataIndex, codeIndex) if err != nil { - return nil, solana.Hash{}, dataIndex, codeIndex, err + return shredPackets{}, dataIndex, codeIndex, err } packets = append(packets, batchPackets...) + fecSetRoots = append(fecSetRoots, root) chainedRoot = root dataIndex += dataShredsPerFECBlock codeIndex += codingShredsPerFECBlock } - return packets, chainedRoot, dataIndex, codeIndex, nil + return shredPackets{ + packets: packets, fecSetRoots: fecSetRoots, chainedMerkleRoot: chainedRoot, + }, dataIndex, codeIndex, nil } func (g *ShredGenerator) makeFECBatch( + encoder reedsolomon.Encoder, leader solana.PrivateKey, data []byte, dataCap int, @@ -142,6 +201,7 @@ func (g *ShredGenerator) makeFECBatch( resigned bool, parentOffset uint16, flags byte, + dataComplete bool, isLastInSlot bool, chainedMerkleRoot solana.Hash, dataIndex uint32, @@ -196,11 +256,11 @@ func (g *ShredGenerator) makeFECBatch( dataPackets[i][dataFlagsOffset] |= shredFlagLastShredInSlot break } - } else if len(dataPackets) > 0 { + } else if dataComplete && len(dataPackets) > 0 { dataPackets[len(dataPackets)-1][dataFlagsOffset] |= shredFlagDataComplete } - root, err := finishErasureBatch(leader, allPackets, chainedMerkleRoot, proofSize, resigned) + root, err := finishErasureBatch(encoder, leader, allPackets, chainedMerkleRoot, proofSize, resigned) if err != nil { return nil, solana.Hash{}, err } @@ -208,101 +268,70 @@ func (g *ShredGenerator) makeFECBatch( } func finishErasureBatch( + encoder reedsolomon.Encoder, leader solana.PrivateKey, packets [][]byte, chainedMerkleRoot solana.Hash, proofSize uint8, resigned bool, ) (solana.Hash, error) { - encoder, err := reedsolomon.New(dataShredsPerFECBlock, codingShredsPerFECBlock) + if len(packets) != dataShredsPerFECBlock+codingShredsPerFECBlock { + return solana.Hash{}, fmt.Errorf("invalid FEC packet count %d", len(packets)) + } + dataCap, err := merkleCapacity(dataPayloadSize, dataHeaderSize, proofSize, true, resigned) if err != nil { return solana.Hash{}, err } + codeCap, err := merkleCapacity(codingPayloadSize, codingHeaderSize, proofSize, true, resigned) + if err != nil { + return solana.Hash{}, err + } + dataVariant := chainedDataVariant(proofSize, resigned) + codeVariant := chainedCodeVariant(proofSize, resigned) + // These packets were constructed immediately above, so retain direct views + // of their erasure regions. ParseShred is intentionally a defensive, + // owning parser for untrusted network packets; using it here would allocate + // and copy every packet several times only to copy the same bytes back. shards := make([][]byte, len(packets)) for i, packet := range packets { - shred, err := ParseShred(packet) - if err != nil { - return solana.Hash{}, fmt.Errorf("parse batch shred %d: %w", i, err) + if i < dataShredsPerFECBlock { + if len(packet) < dataPayloadSize || packet[shredVariantOffset] != dataVariant { + return solana.Hash{}, fmt.Errorf("invalid generated data shred %d", i) + } + shards[i] = packet[shredSignatureSize : dataHeaderSize+dataCap] + continue } - shard, err := shred.erasureShard() - if err != nil { - return solana.Hash{}, fmt.Errorf("erasure shard %d: %w", i, err) + if len(packet) < codingPayloadSize || packet[shredVariantOffset] != codeVariant { + return solana.Hash{}, fmt.Errorf("invalid generated coding shred %d", i-dataShredsPerFECBlock) } - shards[i] = shard + shards[i] = packet[codingHeaderSize : codingHeaderSize+codeCap] } if err := encoder.Encode(shards); err != nil { return solana.Hash{}, fmt.Errorf("reed-solomon encode: %w", err) } for i, packet := range packets { - shred, err := ParseShred(packet) - if err != nil { - return solana.Hash{}, err - } - proofSizeInfo, chained, resignedFlag, ok := merkleVariantInfo(shred.Variant) - if !ok { - return solana.Hash{}, ErrUnsupportedShred - } - _ = proofSizeInfo - _ = chained - _ = resignedFlag - - capacity, err := merkleCapacity(len(packet), dataHeaderSize, proofSize, true, resigned) - if shred.Type == ShredTypeCode { - capacity, err = merkleCapacity(len(packet), codingHeaderSize, proofSize, true, resigned) - } - if err != nil { - return solana.Hash{}, err - } - rootOffset := dataHeaderSize + capacity - if shred.Type == ShredTypeCode { - rootOffset = codingHeaderSize + capacity + rootOffset := dataHeaderSize + dataCap + if i >= dataShredsPerFECBlock { + rootOffset = codingHeaderSize + codeCap } copy(packet[rootOffset:rootOffset+merkleRootSize], chainedMerkleRoot[:]) - - if shred.Type == ShredTypeCode { - start := codingHeaderSize - end := start + capacity - copy(packet[start:end], shards[i]) - } else { - start := shredSignatureSize - end := dataHeaderSize + capacity - copy(packet[start:end], shards[i]) - } } - nodes, err := buildMerkleTree(packets) - if err != nil { - return solana.Hash{}, err - } + nodes := buildGeneratedMerkleTree(packets, dataCap, codeCap) root := nodes[len(nodes)-1] sig := ed25519.Sign(ed25519.PrivateKey(leader), root[:]) - for _, packet := range packets { + for i, packet := range packets { copy(packet[shredSignatureOffset:shredSignatureSize], sig) - shred, err := ParseShred(packet) - if err != nil { - return solana.Hash{}, err - } - leafIndex, err := shred.merkleLeafIndex() - if err != nil { - return solana.Hash{}, err - } - proof := makeMerkleProof(nodes, leafIndex, len(packets)) - capacity, err := merkleCapacity(len(packet), dataHeaderSize, proofSize, true, resigned) - if shred.Type == ShredTypeCode { - capacity, err = merkleCapacity(len(packet), codingHeaderSize, proofSize, true, resigned) - } - if err != nil { - return solana.Hash{}, err - } - proofOffset := dataHeaderSize + capacity + merkleRootSize - if shred.Type == ShredTypeCode { - proofOffset = codingHeaderSize + capacity + merkleRootSize + proofOffset := dataHeaderSize + dataCap + merkleRootSize + if i >= dataShredsPerFECBlock { + proofOffset = codingHeaderSize + codeCap + merkleRootSize } - for j, entry := range proof { - copy(packet[proofOffset+j*merkleProofEntrySize:], entry[:]) + proofEntries := writeMerkleProof(packet[proofOffset:], nodes, i, len(packets)) + if proofEntries != int(proofSize) { + return solana.Hash{}, fmt.Errorf("generated merkle proof has %d entries, want %d", proofEntries, proofSize) } if resigned { retransmitOffset := proofOffset + int(proofSize)*merkleProofEntrySize @@ -312,6 +341,54 @@ func finishErasureBatch( return root, nil } +// buildGeneratedMerkleTree hashes the fixed packet order emitted by +// makeFECBatch: 32 data shreds followed by 32 coding shreds. Callers must have +// already validated the packet sizes and variants in finishErasureBatch. +func buildGeneratedMerkleTree(packets [][]byte, dataCap, codeCap int) []solana.Hash { + leaves := make([]solana.Hash, len(packets)) + for i, packet := range packets { + end := dataHeaderSize + dataCap + merkleRootSize + if i >= dataShredsPerFECBlock { + end = codingHeaderSize + codeCap + merkleRootSize + } + leaves[i] = merkleHashLeaf(packet[shredSignatureSize:end]) + } + + nodes := make([]solana.Hash, 0, merkleTreeSize(len(leaves))) + nodes = append(nodes, leaves...) + for size := len(leaves); size > 1; size = (size + 1) >> 1 { + offset := len(nodes) - size + for index := offset; index < offset+size; index += 2 { + other := index + 1 + if other >= offset+size { + other = offset + size - 1 + } + nodes = append(nodes, merkleHashNode(nodes[index][:merkleProofEntrySize], nodes[other][:merkleProofEntrySize])) + } + } + return nodes +} + +// writeMerkleProof writes the truncated sibling hashes directly into a packet. +// The generated FEC tree has fixed depth, so materializing a temporary proof +// slice for every one of its 64 packets only adds allocator and copy traffic. +func writeMerkleProof(dst []byte, nodes []solana.Hash, index, size int) int { + entries := 0 + offset := 0 + for size > 1 { + sibling := index ^ 1 + if sibling >= size { + sibling = size - 1 + } + copy(dst[entries*merkleProofEntrySize:], nodes[offset+sibling][:merkleProofEntrySize]) + entries++ + offset += size + size = (size + 1) >> 1 + index >>= 1 + } + return entries +} + func buildMerkleTree(packets [][]byte) ([]solana.Hash, error) { leaves := make([]solana.Hash, len(packets)) for i, packet := range packets { diff --git a/pkg/turbine/generate_bench_test.go b/pkg/turbine/generate_bench_test.go new file mode 100644 index 000000000..2301eefaa --- /dev/null +++ b/pkg/turbine/generate_bench_test.go @@ -0,0 +1,242 @@ +package turbine + +import ( + "crypto/ed25519" + "crypto/sha256" + "encoding/hex" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/gagliardetto/solana-go" + "github.com/klauspost/reedsolomon" +) + +const ( + benchmarkTransactionCount = 50_000 + benchmarkTransactionBytes = 1_232 +) + +var ( + benchmarkEncoderSink reedsolomon.Encoder + benchmarkPacketsSink [][]byte + benchmarkRootSink solana.Hash + benchmarkByteSink byte +) + +func benchmarkLeaderKey() solana.PrivateKey { + var seed [ed25519.SeedSize]byte + for i := range seed { + seed[i] = byte(i + 1) + } + return solana.PrivateKey(ed25519.NewKeyFromSeed(seed[:])) +} + +func benchmarkPayload(size int) []byte { + payload := make([]byte, size) + var state uint64 = 0x9e3779b97f4a7c15 + for i := range payload { + // A deterministic, non-zero corpus avoids accidentally benchmarking a + // special all-zero input while keeping fixture construction out of the + // timed region. + state ^= state << 7 + state ^= state >> 9 + state ^= state << 8 + payload[i] = byte(state) + } + return payload +} + +// BenchmarkReedSolomonEncode32x32 isolates the arithmetic kernel used by one +// unsigned 32+32 chained FEC set. Encoder construction, shred parsing, Merkle +// hashing, signing, and packet copies are intentionally outside this result. +func BenchmarkReedSolomonEncode32x32(b *testing.B) { + const shardBytes = 987 // unsigned chained 32+32 shreds with proof size 6 + encoder, err := reedsolomon.New(dataShredsPerFECBlock, codingShredsPerFECBlock) + if err != nil { + b.Fatal(err) + } + shards := make([][]byte, dataShredsPerFECBlock+codingShredsPerFECBlock) + for i := range shards { + shards[i] = make([]byte, shardBytes) + if i < dataShredsPerFECBlock { + copy(shards[i], benchmarkPayload(shardBytes)) + } + } + + b.SetBytes(dataShredsPerFECBlock * shardBytes) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if err := encoder.Encode(shards); err != nil { + b.Fatal(err) + } + } + benchmarkByteSink = shards[len(shards)-1][shardBytes-1] +} + +// BenchmarkReedSolomonNew32x32 measures work that finishErasureBatch currently +// repeats for every FEC set even though the 32+32 shape never changes. +func BenchmarkReedSolomonNew32x32(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + encoder, err := reedsolomon.New(dataShredsPerFECBlock, codingShredsPerFECBlock) + if err != nil { + b.Fatal(err) + } + benchmarkEncoderSink = encoder + } +} + +// BenchmarkMakeShredsFromData reports the complete current generator cost, +// including packet construction, Reed-Solomon coding, chained Merkle trees, +// one Ed25519 signature per FEC set, and proof materialization. +func BenchmarkMakeShredsFromData(b *testing.B) { + const blockBytes = benchmarkTransactionCount * benchmarkTransactionBytes + cases := []struct { + name string + size int + isLastInSlot bool + }{ + {name: "one-unsigned-fec", size: dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false)}, + {name: "block-50000x1232", size: blockBytes, isLastInSlot: true}, + } + + for _, tc := range cases { + b.Run(tc.name, func(b *testing.B) { + leader := benchmarkLeaderKey() + payload := benchmarkPayload(tc.size) + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1} + + b.SetBytes(int64(tc.size)) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + packets, root, _, _, err := gen.MakeShredsFromData(leader, payload, tc.isLastInSlot, solana.Hash{}, 0, 0) + if err != nil { + b.Fatal(err) + } + benchmarkPacketsSink = packets + benchmarkRootSink = root + } + b.StopTimer() + if len(benchmarkPacketsSink) == 0 || len(benchmarkPacketsSink)%64 != 0 { + b.Fatalf("unexpected packet count %d", len(benchmarkPacketsSink)) + } + b.ReportMetric(float64(len(benchmarkPacketsSink)), "packets/op") + b.ReportMetric(float64(len(benchmarkPacketsSink)/64), "FEC-sets/op") + if tc.size == blockBytes { + b.ReportMetric(benchmarkTransactionCount, "transactions/op") + } + }) + } +} + +// BenchmarkMakeShreds50000TargetBatches1232 models the producer's target-sized +// component stream without retaining a multi-gigabyte output. It measures only +// the 61.6 MB transaction payload; entry framing is deliberately outside this +// erasure-coding benchmark. +func BenchmarkMakeShreds50000TargetBatches1232(b *testing.B) { + const inputBytes = benchmarkTransactionCount * benchmarkTransactionBytes + leader := benchmarkLeaderKey() + payload := benchmarkPayload(inputBytes) + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1} + + b.SetBytes(inputBytes) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + var ( + root solana.Hash + dataIndex uint32 + codeIndex uint32 + ) + var ( + offset int + components int + packetCount int + fecSetCount int + ) + for offset < len(payload) { + end := min(offset+costmodel.DefaultTargetBatchBytes, len(payload)) + packets, nextRoot, nextData, nextCode, err := gen.MakeShredsFromData( + leader, + payload[offset:end], + end == len(payload), + root, + dataIndex, + codeIndex, + ) + if err != nil { + b.Fatal(err) + } + benchmarkPacketsSink = packets + components++ + packetCount += len(packets) + fecSetCount += len(packets) / (dataShredsPerFECBlock + codingShredsPerFECBlock) + root, dataIndex, codeIndex = nextRoot, nextData, nextCode + offset = end + } + b.ReportMetric(float64(components), "components/op") + b.ReportMetric(float64(fecSetCount), "FEC-sets/op") + b.ReportMetric(float64(packetCount), "packets/op") + benchmarkRootSink = root + } + b.StopTimer() + b.ReportMetric(benchmarkTransactionCount, "transactions/op") +} + +func TestBenchmarkBlockPayloadAccounting(t *testing.T) { + const blockBytes = benchmarkTransactionCount * benchmarkTransactionBytes + unsignedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + signedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, true) + unsignedBytes := blockBytes - signedBatch + unsignedFECs := (unsignedBytes + unsignedBatch - 1) / unsignedBatch + totalFECs := unsignedFECs + 1 + if totalFECs != 2000 { + t.Fatalf("50k x 1232 payload maps to %d FEC sets, want 2000 (%s)", totalFECs, fmt.Sprintf("%d bytes", blockBytes)) + } +} + +func TestProducerTargetMatchesTwoTypicalFECPayloads(t *testing.T) { + want := 2 * dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + if costmodel.DefaultTargetBatchBytes != want { + t.Fatalf("producer target = %d, want two typical FEC payloads = %d", costmodel.DefaultTargetBatchBytes, want) + } +} + +func TestMakeShredsFromDataStableBytes(t *testing.T) { + unsignedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + signedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, true) + tests := []struct { + name string + size int + isLastInSlot bool + want string + }{ + {name: "unsigned-one-fec", size: unsignedBatch, want: "f9334f1835240df21d4b48a09f35b3ff90578122d30d0527608c10d38d0911f7"}, + {name: "signed-one-fec", size: signedBatch, isLastInSlot: true, want: "bfa398c445509c5e1345553bbe84fea04e86001caf441008b4014d07c6e36ccd"}, + {name: "two-unsigned-one-signed", size: 2*unsignedBatch + signedBatch, isLastInSlot: true, want: "dffae840e4c247680b5e1667747a63138872a0080c51a6fcf4cc5002eb7778ac"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1, ReferenceTick: 17} + packets, _, _, _, err := gen.MakeShredsFromData( + benchmarkLeaderKey(), benchmarkPayload(tt.size), tt.isLastInSlot, + solana.Hash{3}, 7, 11, + ) + if err != nil { + t.Fatal(err) + } + h := sha256.New() + for _, packet := range packets { + _, _ = h.Write(packet) + } + got := hex.EncodeToString(h.Sum(nil)) + if got != tt.want { + t.Fatalf("packet digest %s, want %s", got, tt.want) + } + }) + } +} diff --git a/pkg/turbine/generate_test.go b/pkg/turbine/generate_test.go index 083133954..64c5d1091 100644 --- a/pkg/turbine/generate_test.go +++ b/pkg/turbine/generate_test.go @@ -4,6 +4,7 @@ import ( "bytes" "crypto/ed25519" "encoding/binary" + "fmt" "testing" "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" @@ -174,6 +175,46 @@ func TestMakeShredsFromDataRoundTrip(t *testing.T) { } } +func TestGeneratedFECRootsMatchPacketProofs(t *testing.T) { + unsignedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + signedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, true) + for _, last := range []bool{false, true} { + for _, size := range []int{0, 1, signedBatch, signedBatch + 1, unsignedBatch, unsignedBatch + 1, 2 * unsignedBatch, 2*unsignedBatch + signedBatch} { + t.Run(fmt.Sprintf("last=%t/bytes=%d", last, size), func(t *testing.T) { + gen := ShredGenerator{Slot: 100, ParentSlot: 99, Version: 7, ReferenceTick: 17} + parentRoot := solana.Hash{5} + leader := testShredLeader(t) + batch, nextData, nextCode, err := gen.makeShredsFromData( + leader, benchmarkPayload(size), last, parentRoot, 7, 11, + ) + require.NoError(t, err) + require.NotEmpty(t, batch.fecSetRoots) + require.Len(t, batch.packets, len(batch.fecSetRoots)*(dataShredsPerFECBlock+codingShredsPerFECBlock)) + require.Equal(t, uint32(7+len(batch.fecSetRoots)*dataShredsPerFECBlock), nextData) + require.Equal(t, uint32(11+len(batch.fecSetRoots)*codingShredsPerFECBlock), nextCode) + require.Equal(t, batch.fecSetRoots[len(batch.fecSetRoots)-1], batch.chainedMerkleRoot) + for i, packet := range batch.packets { + fec := i / (dataShredsPerFECBlock + codingShredsPerFECBlock) + shred, err := ParseShred(packet) + require.NoError(t, err) + root, err := shred.MerkleRoot() + require.NoError(t, err) + require.Equal(t, batch.fecSetRoots[fec], root) + require.Equal(t, uint32(7+fec*dataShredsPerFECBlock), shred.FECSetIndex) + previousRoot := parentRoot + if fec > 0 { + previousRoot = batch.fecSetRoots[fec-1] + } + embeddedRoot, err := shred.EmbeddedChainedMerkleRoot() + require.NoError(t, err) + require.Equal(t, previousRoot, embeddedRoot) + require.NoError(t, shred.VerifySignature(leader.PublicKey())) + } + }) + } + } +} + func TestMakeShredsFromAlpenglowBlock(t *testing.T) { leader := testShredLeader(t) gen := ShredGenerator{ @@ -184,10 +225,10 @@ func TestMakeShredsFromAlpenglowBlock(t *testing.T) { } var ( - chainedRoot = solana.Hash{5} - nextData uint32 = 0 - nextCode uint32 = 0 - allDataShreds []*Shred + chainedRoot = solana.Hash{5} + nextData uint32 = 0 + nextCode uint32 = 0 + allDataShreds []*Shred ) for _, component := range buildAlpenglowSlot(t) { packets, root, newData, newCode, err := gen.MakeShredsFromData( diff --git a/pkg/turbine/generated_component_boundary_test.go b/pkg/turbine/generated_component_boundary_test.go new file mode 100644 index 000000000..871843089 --- /dev/null +++ b/pkg/turbine/generated_component_boundary_test.go @@ -0,0 +1,47 @@ +package turbine + +import ( + "bytes" + "fmt" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Exercise exact FEC boundaries and the smaller resigned final-set capacity. +// The receiver must see one DATA_COMPLETE per serialized component, regardless +// of how many FEC sets carry it, with proofs authenticating the final flags. +func TestGeneratedComponentHasOneCompletionBoundary(t *testing.T) { + unsigned := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + signed := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, true) + for _, last := range []bool{false, true} { + for _, size := range []int{0, 1, signed, signed + 1, unsigned, unsigned + 1, 2 * unsigned, 2*unsigned + signed} { + t.Run(fmt.Sprintf("last=%t/bytes=%d", last, size), func(t *testing.T) { + leader := testShredLeader(t) + gen := ShredGenerator{Slot: 100, ParentSlot: 99, Version: 7} + payload := bytes.Repeat([]byte{0x5a}, size) + packets, _, nextData, _, err := gen.MakeShredsFromData(leader, payload, last, solana.Hash{}, 0, 0) + require.NoError(t, err) + var decoded []byte + var completed int + for _, packet := range packets { + shred, err := ParseShred(packet) + require.NoError(t, err) + require.NoError(t, shred.VerifySignature(leader.PublicKey())) + if shred.Type != ShredTypeData { + continue + } + decoded = append(decoded, shred.Data...) + if shred.DataComplete() { + completed++ + require.Equal(t, nextData-1, shred.Index) + } + require.Equal(t, last && shred.Index == nextData-1, shred.LastInSlot()) + } + require.Equal(t, 1, completed) + require.True(t, bytes.Equal(payload, decoded)) + }) + } + } +} diff --git a/pkg/turbine/internal/rsrecover/doc.go b/pkg/turbine/internal/rsrecover/doc.go new file mode 100644 index 000000000..38bc6e7f5 --- /dev/null +++ b/pkg/turbine/internal/rsrecover/doc.go @@ -0,0 +1,5 @@ +// Package rsrecover contains fixed-shape Reed-Solomon recovery plans for +// Solana's 32 data + 32 coding shred FEC sets. Production dispatch uses only +// exactly-one-missing-data recovery. Reduced-subset and all-coding plans are +// reference/benchmark alternatives and are not selected by SlotAssembler. +package rsrecover diff --git a/pkg/turbine/internal/rsrecover/recover.go b/pkg/turbine/internal/rsrecover/recover.go new file mode 100644 index 000000000..6a16e91e9 --- /dev/null +++ b/pkg/turbine/internal/rsrecover/recover.go @@ -0,0 +1,599 @@ +package rsrecover + +import ( + "errors" + "fmt" + "math/bits" + "sync" + + "github.com/klauspost/reedsolomon" +) + +const ( + DataShards = 32 + CodingShards = 32 + TotalShards = DataShards + CodingShards + + cauchyGamma = byte(0xa5) +) + +var ( + ErrInvalidPattern = errors.New("invalid erasure pattern") + ErrPatternChanged = errors.New("availability differs from prepared plan") + ErrInvalidBuffers = errors.New("invalid recovery buffers") + + gfLog [256]byte + gfExp [512]byte + + allCodingEncoderOnce sync.Once + allCodingEncoder reedsolomon.Encoder + allCodingEncoderErr error + + // oneDataCoefficientRows[missing][coding] contains one coefficient for + // each data input followed by the selected coding input. Every one of the + // fixed 32x32 rows is exhaustively differential-tested against the general + // Reed-Solomon decoder. + oneDataCoefficientRows [DataShards][CodingShards][DataShards + 1]byte +) + +func init() { + value := uint16(1) + for exponent := 0; exponent < 255; exponent++ { + gfExp[exponent] = byte(value) + gfLog[byte(value)] = byte(exponent) + value <<= 1 + if value&0x100 != 0 { + value ^= 0x11d + } + } + for exponent := 255; exponent < len(gfExp); exponent++ { + gfExp[exponent] = gfExp[exponent-255] + } + for missing := 0; missing < DataShards; missing++ { + for coding := 0; coding < CodingShards; coding++ { + // If c = sum(a_i*d_i), then the missing d_m is + // inv(a_m) * (c + sum(i != m, a_i*d_i)) over GF(2^8). + coefficientInv := reedsolomon.Inv(codingCoefficient(coding, missing)) + for data := 0; data < DataShards; data++ { + if data != missing { + oneDataCoefficientRows[missing][coding][data] = gfMul( + coefficientInv, + codingCoefficient(coding, data), + ) + } + } + oneDataCoefficientRows[missing][coding][DataShards] = coefficientInv + } + } +} + +// OneDataPlan recovers exactly one absent data shard from the other 31 data +// shards and one available coding shard. It avoids a general 32x32 decode +// matrix inversion. The plan is immutable and safe for concurrent execution +// when callers provide independent destinations. +type OneDataPlan struct { + presence uint64 + missing uint8 + sources [DataShards]uint8 + coefficients [DataShards]byte +} + +// DataSubsetPlan recovers every absent data shard using the present data and +// the lowest-indexed coding rows required to reach the 32-shard threshold. +// Setup solves only the m x m system induced by the missing data columns. +type DataSubsetPlan struct { + presence uint64 + missing []uint8 + sources [DataShards]uint8 + weights [][]byte +} + +// AllCodingPlan recovers all 32 data rows from all 32 coding rows. For the +// fixed Solana matrix C*C=I, so the package's optimized encoder can apply C a +// second time instead of constructing a decode matrix. +type AllCodingPlan struct { + presence uint64 + encoder reedsolomon.Encoder +} + +// Presence reports which of the 64 input shards are non-empty. A zero-length +// shard is absent, matching reedsolomon.ReconstructSome. +func Presence(shards [][]byte) (uint64, error) { + if len(shards) != TotalShards { + return 0, fmt.Errorf("%w: got %d shards, want %d", ErrInvalidBuffers, len(shards), TotalShards) + } + var mask uint64 + for index, shard := range shards { + if len(shard) != 0 { + mask |= uint64(1) << index + } + } + return mask, nil +} + +// PrepareRecoverOneData constructs the direct coefficient row for a pattern +// with exactly one missing data shard. Additional coding shards may be present; +// the lowest-indexed one is selected deterministically. +func PrepareRecoverOneData(presence uint64, missingDataIndex int) (OneDataPlan, error) { + if missingDataIndex < 0 || missingDataIndex >= DataShards { + return OneDataPlan{}, fmt.Errorf("%w: missing data index %d", ErrInvalidPattern, missingDataIndex) + } + for index := 0; index < DataShards; index++ { + present := presence&(uint64(1)<= DataShards { + return fmt.Errorf("%w: missing data index %d", ErrInvalidPattern, missingDataIndex) + } + const dataMask = uint64(1)<> DataShards) + if codingMask == 0 { + return fmt.Errorf("%w: no coding shard is available", ErrInvalidPattern) + } + shardSize, err := validateExecution(presence, shards, [][]byte{dst}) + if err != nil { + return err + } + if len(dst) != shardSize { + return fmt.Errorf("%w: destination has %d bytes, want %d", ErrInvalidBuffers, len(dst), shardSize) + } + + codingPosition := bits.TrailingZeros32(codingMask) + row := &oneDataCoefficientRows[missingDataIndex][codingPosition] + var lowLevel reedsolomon.LowLevel + first := true + for dataIndex := 0; dataIndex < DataShards; dataIndex++ { + coefficient := row[dataIndex] + if coefficient == 0 { + continue + } + if first { + lowLevel.GalMulSlice(coefficient, shards[dataIndex], dst) + first = false + } else { + lowLevel.GalMulSliceXor(coefficient, shards[dataIndex], dst) + } + } + lowLevel.GalMulSliceXor(row[DataShards], shards[DataShards+codingPosition], dst) + return nil +} + +// PrepareRecoverDataSubset constructs direct output rows for all absent data +// shards. It first inverts only the reduced m x m coding/data matrix, then +// expands those rows over exactly 32 selected input shards so byte execution +// can be compared fairly with a general decoder. +func PrepareRecoverDataSubset(presence uint64) (DataSubsetPlan, error) { + plan := DataSubsetPlan{presence: presence} + knownData := make([]uint8, 0, DataShards) + for index := 0; index < DataShards; index++ { + if presence&(uint64(1)< code.Position { + code = s + } + } + if code == nil { + return fmt.Errorf("recover FEC: missing coding template") + } + proofSize, chained, _, ok := merkleVariantInfo(code.Variant) + if !ok || int(proofSize) != bits.Len(uint(len(shards)-1)) { + return fmt.Errorf("recover FEC: invalid Merkle proof depth") + } + if code.Index < uint32(code.Position) { + return fmt.Errorf("recover FEC: invalid coding index") + } + codeBase := code.Index - uint32(code.Position) + if uint64(codeBase)+uint64(f.layout.codingShreds)-1 > math.MaxUint32 { + return fmt.Errorf("recover FEC: coding index overflow") + } + expected, err := code.MerkleRoot() + if err != nil { + return err + } + encoder, err := a.fecEncoder(f.layout) + if err != nil { + return err + } + // Present shards are read-only. Only absent coding shards remain after the + // caller's data recovery, so this also checks commitments to missing parity. + if err = encoder.Reconstruct(shards); err != nil { + return fmt.Errorf("recover FEC authentication: %w", err) + } + dataCount := int(f.layout.dataShreds) + data := make(map[uint32]*Shred, len(recovered)) + for _, s := range recovered { + // Chained root and retransmitter signature are outside the erasure region. + // Copy the template suffix now, then replace its proof after authenticating. + copy(s.Payload[shredSignatureSize+f.layout.shardSize:], code.Payload[codingHeaderSize+f.layout.shardSize:]) + data[s.Index-f.fecSetIndex] = s + } + nodes := make([]solana.Hash, 0, merkleTreeSize(len(shards))) + for i, shard := range shards { + var s *Shred + if i < dataCount { + s = f.data[uint32(i)] + if s == nil { + s = data[uint32(i)] + } + } else { + pos := uint16(i - dataCount) + s = f.coding[pos] + if s == nil { + payload := append([]byte(nil), code.Payload...) + binary.LittleEndian.PutUint32(payload[shredIndexOffset:], codeBase+uint32(pos)) + binary.LittleEndian.PutUint16(payload[codingPositionOffset:], pos) + copy(payload[codingHeaderSize:], shard) + // Only header fields and bytes consumed by merkleLeaf are needed here. + copyOfCode := *code + copyOfCode.Payload = payload + copyOfCode.Index = codeBase + uint32(pos) + copyOfCode.Position = pos + s = ©OfCode + } + } + if s == nil { + return fmt.Errorf("recover FEC: missing reconstructed data %d", i) + } + leaf, err := s.merkleLeaf() + if err != nil { + return err + } + nodes = append(nodes, leaf) + } + for size := len(shards); size > 1; size = (size + 1) >> 1 { + offset := len(nodes) - size + for i := 0; i < size; i += 2 { + right := min(i+1, size-1) + nodes = append(nodes, merkleHashNode(nodes[offset+i][:merkleProofEntrySize], nodes[offset+right][:merkleProofEntrySize])) + } + } + if nodes[len(nodes)-1] != expected { + return fmt.Errorf("%w: recovered FEC Merkle root mismatch slot=%d fec_set=%d", ErrInvalidSignature, f.slot, f.fecSetIndex) + } + proofOffset := shredSignatureSize + f.layout.shardSize + if chained { + proofOffset += merkleRootSize + } + for _, s := range recovered { + writeMerkleProof(s.Payload[proofOffset:], nodes, int(s.Index-f.fecSetIndex), len(shards)) + } + return nil +} diff --git a/pkg/turbine/recovery_authentication_test.go b/pkg/turbine/recovery_authentication_test.go new file mode 100644 index 000000000..626742850 --- /dev/null +++ b/pkg/turbine/recovery_authentication_test.go @@ -0,0 +1,226 @@ +package turbine + +import ( + "crypto/ed25519" + "encoding/binary" + "fmt" + "sort" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/klauspost/reedsolomon" + "github.com/stretchr/testify/require" +) + +func resignRecoveryFixture(t *testing.T, packets [][]byte) { + t.Helper() + nodes, err := buildMerkleTree(packets) + require.NoError(t, err) + root := nodes[len(nodes)-1] + sig := ed25519.Sign(ed25519.PrivateKey(benchmarkLeaderKey()), root[:]) + for i, p := range packets { + s, err := ParseShred(p) + require.NoError(t, err) + shard, err := s.erasureShard() + require.NoError(t, err) + start := shredSignatureSize + if s.Type == ShredTypeCode { + start = codingHeaderSize + } + _, chained, _, _ := merkleVariantInfo(s.Variant) + offset := start + len(shard) + if chained { + offset += merkleRootSize + } + copy(p[:shredSignatureSize], sig) + writeMerkleProof(p[offset:], nodes, i, len(packets)) + } +} + +func recoveryFixtureState(t testing.TB, packets [][]byte, missing int) (*SlotAssembler, *slotState) { + t.Helper() + state := &slotState{slot: 10, shreds: make(map[uint32]*Shred), fecSets: make(map[uint32]*fecState), lastIndex: ^uint32(0)} + // Exactly 32 received shreds: also force reconstruction of missing parity. + for i, p := range packets { + if i < missing || i >= 32+missing { + continue + } + s, err := ParseShred(p) + require.NoError(t, err) + state.slot = s.Slot + state.shredVer = s.Version + if s.Type == ShredTypeData { + require.NoError(t, state.addDataShred(s)) + } else { + require.NoError(t, state.addCodingShred(s)) + } + } + return NewSlotAssembler(), state +} + +func TestRecoveredFECAuthentication(t *testing.T) { + for _, resigned := range []bool{false, true} { + for _, missing := range []int{1, 3, 32} { + for _, attack := range []string{"valid", "received_parity", "missing_parity", "recovered_bytes"} { + t.Run(fmt.Sprintf("resigned=%t/missing=%d/%s", resigned, missing, attack), func(t *testing.T) { + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1} + packets, _, _, _, err := gen.MakeShredsFromData(benchmarkLeaderKey(), benchmarkPayload(128), resigned, solana.Hash{9}, 0, 0) + require.NoError(t, err) + require.Len(t, packets, 64) + if attack == "received_parity" { + packets[32][codingHeaderSize+128] ^= 1 + resignRecoveryFixture(t, packets) + } + if attack == "missing_parity" { + if missing == 32 { + t.Skip("no missing coding shard") + } + packets[63][codingHeaderSize+128] ^= 1 + resignRecoveryFixture(t, packets) + } + // The adversarial fixtures have valid leader signatures, not corrupted + // network packets: only RS/tree consistency is wrong. + for _, p := range packets { + s, e := ParseShred(p) + require.NoError(t, e) + require.NoError(t, s.VerifySignature(benchmarkLeaderKey().PublicKey())) + } + a, state := recoveryFixtureState(t, packets, missing) + recovered, err := a.recoverFEC(state, 0) + if attack == "received_parity" || attack == "missing_parity" { + require.ErrorIs(t, err, ErrInvalidSignature) + require.Empty(t, recovered) + return + } + require.NoError(t, err) + require.Len(t, recovered, missing) + for _, s := range recovered { + require.Equal(t, packets[s.Index], s.Payload) + require.NoError(t, s.VerifySignature(benchmarkLeaderKey().PublicKey())) + } + if attack == "recovered_bytes" { + f := state.fecSets[0] + shards := make([][]byte, 64) + for i, s := range f.data { + shards[i], err = s.erasureShard() + require.NoError(t, err) + } + for i, s := range f.coding { + shards[32+int(i)], err = s.erasureShard() + require.NoError(t, err) + } + for _, s := range recovered { + shards[s.Index], err = s.erasureShard() + require.NoError(t, err) + } + recovered[0].Payload[dataHeaderSize+1] ^= 1 + require.ErrorIs(t, a.authenticateRecoveredFEC(f, shards, recovered), ErrInvalidSignature) + } + }) + } + } + } +} + +// Recreate coding packets from immutable Agave data packets, without signing a +// new root. Equality to the root committed by Agave checks erasure layout, +// parity coefficients, coding headers, chain roots and Merkle construction. +func TestRecoveredFECAgaveSignedCapture(t *testing.T) { + all := agavePaddedSlot1752420Packets(t) + sort.Slice(all, func(i, j int) bool { + return binary.LittleEndian.Uint32(all[i][shredIndexOffset:]) < binary.LittleEndian.Uint32(all[j][shredIndexOffset:]) + }) + for group := 0; group < 4; group++ { + packets := make([][]byte, 64) + shards := make([][]byte, 64) + var first *Shred + for i := 0; i < 32; i++ { + // Captured repair packets may append a four-byte nonce, which is + // transport metadata rather than part of the authenticated shred. + packets[i] = all[group*32+i][:dataPayloadSize] + s, err := ParseShred(packets[i]) + require.NoError(t, err) + if i == 0 { + first = s + } + shards[i], err = s.erasureShard() + require.NoError(t, err) + } + codeVariant, ok := merkleCounterpartVariant(first.Variant, ShredTypeCode) + require.True(t, ok) + for i := 0; i < 32; i++ { + p := make([]byte, codingPayloadSize) + copy(p[:codingNumDataOffset], first.Payload[:codingNumDataOffset]) + p[shredVariantOffset] = codeVariant + binary.LittleEndian.PutUint32(p[shredIndexOffset:], first.FECSetIndex+uint32(i)) + binary.LittleEndian.PutUint16(p[codingNumDataOffset:], 32) + binary.LittleEndian.PutUint16(p[codingNumCodingOffset:], 32) + binary.LittleEndian.PutUint16(p[codingPositionOffset:], uint16(i)) + copy(p[codingHeaderSize+len(shards[0]):], first.Payload[shredSignatureSize+len(shards[0]):]) + packets[32+i] = p + shards[32+i] = p[codingHeaderSize : codingHeaderSize+len(shards[0])] + } + enc, err := reedsolomon.New(32, 32) + require.NoError(t, err) + require.NoError(t, enc.Encode(shards)) + nodes, err := buildMerkleTree(packets) + require.NoError(t, err) + root, err := first.MerkleRoot() + require.NoError(t, err) + require.Equal(t, root, nodes[len(nodes)-1]) + proofSize, chained, _, _ := merkleVariantInfo(codeVariant) + offset := codingHeaderSize + len(shards[0]) + if chained { + offset += merkleRootSize + } + for i := 32; i < 64; i++ { + require.Equal(t, int(proofSize), writeMerkleProof(packets[i][offset:], nodes, i, 64)) + } + for _, missing := range []int{1, 3, 32} { + a, state := recoveryFixtureState(t, packets, missing) + got, err := a.recoverFEC(state, first.FECSetIndex) + require.NoError(t, err) + require.Len(t, got, missing) + for _, s := range got { + end := dataPayloadSize + _, _, resigned, _ := merkleVariantInfo(s.Variant) + if resigned { + // Hop signatures can differ by relay; Agave copies the + // received coding template's signature, not a lost one. + end -= shredSignatureSize + want, err := first.RetransmitterSignature() + require.NoError(t, err) + got, err := s.RetransmitterSignature() + require.NoError(t, err) + require.Equal(t, want, got) + } + require.Equal(t, packets[s.Index-first.FECSetIndex][:end], s.Payload[:end]) + actual, err := s.MerkleRoot() + require.NoError(t, err) + require.Equal(t, root, actual) + } + } + } +} + +func BenchmarkAuthenticatedFECRecovery(b *testing.B) { + for _, missing := range []int{1, 3, 32} { + b.Run(fmt.Sprintf("missing_data=%d", missing), func(b *testing.B) { + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1} + packets, _, _, _, err := gen.MakeShredsFromData(benchmarkLeaderKey(), benchmarkPayload(128), false, solana.Hash{9}, 0, 0) + require.NoError(b, err) + a, state := recoveryFixtureState(b, packets, missing) + _, err = a.recoverFEC(state, 0) + require.NoError(b, err) + b.ReportAllocs() + b.ResetTimer() + for b.Loop() { + recovered, err := a.recoverFEC(state, 0) + if err != nil { + b.Fatal(err) + } + benchmarkRecoveredShredsSink = recovered + } + }) + } +} diff --git a/pkg/turbine/repair.go b/pkg/turbine/repair.go index f9be294af..7e19baa1d 100644 --- a/pkg/turbine/repair.go +++ b/pkg/turbine/repair.go @@ -304,8 +304,9 @@ type RepairPeerReport struct { } type repairClient struct { - identity ed25519.PrivateKey - peerSource RepairPeerSource + priorityWake chan struct{} // initialized before the receiver starts; coalesced hints + identity ed25519.PrivateKey + peerSource RepairPeerSource mu sync.Mutex outstanding map[repairRequestKey]outstandingRepairRequest @@ -375,28 +376,77 @@ func newRepairClient(identity ed25519.PrivateKey, peerSource RepairPeerSource) ( return nil, fmt.Errorf("repair peer source is required") } c := &repairClient{ - identity: append(ed25519.PrivateKey(nil), identity...), - peerSource: peerSource, - outstanding: make(map[repairRequestKey]outstandingRepairRequest), - byResponse: make(map[repairResponseKey]repairRequestKey), - inflight: make(map[shredKey]*shredInflight), - perPeer: make(map[repairAddressKey]*peerRecord), - expiredCur: make(map[repairResponseKey]outstandingRepairRequest, repairExpiredGenMin), + priorityWake: make(chan struct{}, 1), + identity: append(ed25519.PrivateKey(nil), identity...), + peerSource: peerSource, + outstanding: make(map[repairRequestKey]outstandingRepairRequest), + byResponse: make(map[repairResponseKey]repairRequestKey), + inflight: make(map[shredKey]*shredInflight), + perPeer: make(map[repairAddressKey]*peerRecord), + expiredCur: make(map[repairResponseKey]outstandingRepairRequest, repairExpiredGenMin), } c.timeoutNanos.Store(int64(repairMinRequestTimeout)) return c, nil } +// wakePriority does not send requests or mint rate tokens. It only asks the +// single repair loop to reconsider newly prioritized work sooner. +func (c *repairClient) wakePriority() { + select { + case c.priorityWake <- struct{}{}: + default: + } +} + func (c *repairClient) run(ctx context.Context, conn *net.UDPConn, assembler *SlotAssembler) { - ticker := time.NewTicker(repairScanInterval) + runRepairSchedule(ctx, c.priorityWake, repairScanInterval, 20*time.Millisecond, func() { + c.expireOutstanding(time.Now()) + c.repairOnce(conn, assembler) + }) +} + +// Keep periodic scans for retries/freshness. Coalesce urgent hints and bound +// scan frequency; all sends still use the existing token bucket, admission, +// retry, fanout and peer budgets. The loop remains the sole sender. +func runRepairSchedule(ctx context.Context, wake <-chan struct{}, interval, minSpacing time.Duration, scan func()) { + ticker := time.NewTicker(interval) defer ticker.Stop() + var timer *time.Timer + var urgent <-chan time.Time + var last time.Time + stopTimer := func() { + if timer != nil { + timer.Stop() + } + urgent = nil + } + defer stopTimer() + run := func() { + stopTimer() + if ctx.Err() != nil { + return + } + scan() + last = time.Now() + } + schedule := func() { + if delay := minSpacing - time.Since(last); delay <= 0 { + run() + } else if urgent == nil { + timer = time.NewTimer(delay) + urgent = timer.C + } + } for { select { case <-ctx.Done(): return case <-ticker.C: - c.expireOutstanding(time.Now()) - c.repairOnce(conn, assembler) + schedule() + case <-urgent: + run() + case <-wake: + schedule() } } } @@ -407,7 +457,8 @@ func (c *repairClient) repairOnce(conn *net.UDPConn, assembler *SlotAssembler) { return } priority, edge := assembler.RepairRequestsTiered(repairMaxSlotsPerScan, repairMaxMissingPerSlot) - if len(priority)+len(edge) == 0 { + child, haveChild := assembler.childRepairRequest(time.Now()) + if len(priority)+len(edge) == 0 && !haveChild { return } @@ -433,6 +484,9 @@ func (c *repairClient) repairOnce(conn *net.UDPConn, assembler *SlotAssembler) { } edgeDemand := tierSendDemand(edge, 1) want := tierSendDemand(priority, headInitial) + edgeDemand + if haveChild { + want += len(child.MissingDataShreds) + } if want > repairMaxOutstanding { want = repairMaxOutstanding } @@ -456,6 +510,10 @@ func (c *repairClient) repairOnce(conn *net.UDPConn, assembler *SlotAssembler) { // below what the edge can use; leftover head budget flows to the edge. spent := c.sendTier(conn, peers, priority, splitRepairBudget(budget, edgeDemand), headPol, acct) spent += c.sendTier(conn, peers, edge, budget-spent, nil, acct) + // Parent/normal repair and freshness keep first claim on every token. + if haveChild { + spent += c.sendChildRepair(conn, peers, child, budget-spent, acct) + } c.returnRateTokens(budget - spent) } @@ -563,21 +621,23 @@ func shredSatisfiesRequest(key repairRequestKey, shred *Shred) bool { } } -// observeShredResponse matches an incoming packet against outstanding repair +// matchShredResponse matches an incoming packet against outstanding repair // requests (responder address + nonce). Returns true when the shred was // delivered BY REPAIR — it answers one of our requests — so the caller can -// attribute it in per-slot repair accounting. -func (c *repairClient) observeShredResponse(conn *net.UDPConn, packet []byte, from *net.UDPAddr, shred *Shred) bool { +// attribute it in per-slot repair accounting. highest is a discovery hint for +// followup selection only AFTER the receiver admits the shred; matching and +// peer credit alone do not authorize requests for the claimed range. +func (c *repairClient) matchShredResponse(packet []byte, from *net.UDPAddr, shred *Shred) (matched, highest bool) { if from == nil || shred == nil { - return false + return false, false } nonce, ok := repairproto.ResponseNonce(packet) if !ok { - return false + return false, false } addrKey, ok := repairAddressKeyFromUDP(from) if !ok { - return false + return false, false } responseKey := repairResponseKey{addr: addrKey, nonce: nonce} @@ -611,7 +671,7 @@ func (c *repairClient) observeShredResponse(conn *net.UDPConn, packet []byte, fr // so it still expires into a deserved timeout and keeps its in-flight // slot for retry. The peer gets nothing. c.mu.Unlock() - return false + return false, false } if late { // The expired record lives in exactly one generation; deleting from @@ -641,68 +701,15 @@ func (c *repairClient) observeShredResponse(conn *net.UDPConn, packet []byte, fr c.observeLatencyLocked(latency) c.mu.Unlock() + if entryTraceSelected(shred.Slot) { + traceRepairResponse(outstanding, shred, from.String(), late) + } if late { c.lateResponses.Add(1) } else { c.responses.Add(1) } - // shredSatisfiesRequest already guaranteed slot match and a data shred; the - // gap-backfill path below is HWI-only. - if outstanding.key.kind != repairRequestHighestWindowIndex { - return true - } - - peers := c.peerSnapshot(time.Now()) - if len(peers) == 0 { - return true - } - start := outstanding.key.index - gap := 0 - if shred.Index > start { - gap = int(shred.Index - start) - } - ask := gap - if ask > repairMaxFollowupRequests { - ask = repairMaxFollowupRequests - } - chainProbe := !shred.LastInSlot() && shred.Index < maxDataShredsPerSlot-1 - if chainProbe { - ask++ - } - if ask == 0 { - return true - } - // Followups draw from the SAME token bucket as the scan. This path used - // to be unmetered — with hundreds of probed slots it pushed the total - // send rate ~70% past the cap, which is exactly the flood the peer-side - // QoS ban punishes. When the bucket is dry the scan's deficit-aware - // selection covers the slot on its own cadence. - grant := c.takeRateTokens(ask) - if grant <= 0 { - return true - } - windowBudget := grant - if chainProbe && windowBudget > 0 { - windowBudget-- // reserve the chained probe's token - } - // Discovery followups are bulk-paced and go through the same inflight - // dedup as the scan, so a window index already being repaired is not - // re-sent here. - bulk := bulkPolicy() - acct := c.accountingTimeout() - followups := 0 - for index := start; index < shred.Index && followups < windowBudget; index++ { - if c.sendShredAttempt(conn, peers, repairRequestWindowIndex, shred.Slot, index, bulk, acct) { - followups++ - } - } - if chainProbe && followups < grant { - if c.sendShredAttempt(conn, peers, repairRequestHighestWindowIndex, shred.Slot, shred.Index+1, bulk, acct) { - followups++ - } - } - c.returnRateTokens(grant - followups) - return true + return true, outstanding.key.kind == repairRequestHighestWindowIndex } // observeLatencyLocked folds one request->response latency into the EWMA and @@ -760,7 +767,7 @@ func bulkPolicy() retryPolicy { // satisfyDataShred retires WindowIndex requests satisfied by a verified data // shred arriving through any path: a matched repair response, Turbine // broadcast, FEC/spool hydration, or a duplicate response. Request nonce -// matching still happens first in observeShredResponse so the answering peer +// matching still happens first in matchShredResponse so the answering peer // receives its proper timely/late credit. func (c *repairClient) satisfyDataShred(shred *Shred) { if c == nil || shred == nil || shred.Type != ShredTypeData { @@ -889,7 +896,7 @@ func (c *repairClient) sendShredAttempt(conn *net.UDPConn, peers []gossip.Repair // count), then release BEFORE signing. Ed25519 signing is ~tens of // microseconds; at tens of thousands of req/s, holding the lock across it // serialized every send against the response-processing path - // (observeShredResponse needs the same lock) and could stall the UDP + // (matchShredResponse needs the same lock) and could stall the UDP // receive loop into kernel drops. The reserve is enough for coherence: a // response for this attempt cannot arrive until after we WriteToUDP below, // which is strictly after we register outstanding/byResponse. @@ -933,7 +940,14 @@ func (c *repairClient) sendShredAttempt(conn *net.UDPConn, peers []gossip.Repair c.byResponse[responseKey] = key c.mu.Unlock() + var traceStart int64 + var traceBinding entryRepairTrace + if entryTraceSelected(slot) { + traceStart = entryTraceNow() + traceBinding = entryRepairTrace{Peer: peer.Addr.String(), Nonce: nonce} + } if _, err := conn.WriteToUDP(packet, peer.Addr); err != nil { + traceRepairSend(slot, index, kind, attempt, traceStart, false, traceBinding) c.mu.Lock() delete(c.outstanding, key) delete(c.byResponse, responseKey) @@ -944,6 +958,7 @@ func (c *repairClient) sendShredAttempt(conn *net.UDPConn, peers []gossip.Repair return false } + traceRepairSend(slot, index, kind, attempt, traceStart, true, traceBinding) c.requests.Add(1) return true } @@ -1506,3 +1521,19 @@ func (c *repairClient) stats() RepairStats { AvgResponseMillis: avgResponseMillis, } } + +// Called after normal priority and freshness work. Existing in-flight child +// requests count against lookahead capacity; no fanout or highest-index probes. +func (c *repairClient) sendChildRepair(conn *net.UDPConn, peers []gossip.RepairPeer, req SlotRepairRequest, budget int, acct time.Duration) int { + if budget <= 0 { + return 0 + } + c.mu.Lock() + room := childRepairLimit - c.outstandingForSlotLocked(req.Slot) + c.mu.Unlock() + if room <= 0 { + return 0 + } + req.NeedHighestDataShred = false + return c.sendTier(conn, peers, []SlotRepairRequest{req}, min(room, budget), nil, acct) +} diff --git a/pkg/turbine/repair_followup.go b/pkg/turbine/repair_followup.go new file mode 100644 index 000000000..659ad728d --- /dev/null +++ b/pkg/turbine/repair_followup.go @@ -0,0 +1,73 @@ +package turbine + +import ( + "net" + "time" +) + +// highestRepairFollowup snapshots current deficits AFTER response admission and +// FEC recovery. It never treats a highest-index response as proof that earlier +// pieces are missing. The returned selection can race with later arrivals, but +// no assembler lock is held during signing, network sends, or repair locking. +func (a *SlotAssembler) highestRepairFollowup(slot uint64) (SlotRepairRequest, bool) { + a.mu.Lock() + defer a.mu.Unlock() + if a.slotTooOldLocked(slot) { + return SlotRepairRequest{}, false + } + if _, done := a.completedSlots[slot]; done { + return SlotRepairRequest{}, false + } + s := a.slots[slot] + if s == nil || s.completing { + return SlotRepairRequest{}, false + } + return s.repairRequest(repairMaxFollowupRequests) +} + +// Only invoked for a matched highest-index response which passed admission. +// Disk-only catchup slots defer selection until hydration supplies assembler +// state; blindly backfilling those would ignore data already held in the spool. +func (c *repairClient) followupHighestResponse(conn *net.UDPConn, a *SlotAssembler, slot uint64) { + req, ok := a.highestRepairFollowup(slot) + if !ok { + return + } + peers := c.peerSnapshot(time.Now()) + if len(peers) == 0 { + return + } + // A matched discovery response can request at most 256 missing pieces plus + // one highest-index probe. This response-driven burst shares the global + // token bucket (not the periodic head-share quota); inflight dedup suppresses + // repeat sends and unused tokens are returned below. + ask := len(req.MissingDataShreds) + if req.NeedHighestDataShred { + ask++ + } + grant := c.takeRateTokens(ask) + if grant <= 0 { + return + } + // Reserve discovery capacity as before, even when missing data fills the cap. + window := grant + if req.NeedHighestDataShred { + window-- + } + pol, acct := bulkPolicy(), c.accountingTimeout() + sent := 0 + for _, index := range req.MissingDataShreds { + if sent >= window { + break + } + if c.sendShredAttempt(conn, peers, repairRequestWindowIndex, slot, index, pol, acct) { + sent++ + } + } + if req.NeedHighestDataShred && sent < grant { + if c.sendShredAttempt(conn, peers, repairRequestHighestWindowIndex, slot, req.HighestDataShredIndex, pol, acct) { + sent++ + } + } + c.returnRateTokens(grant - sent) +} diff --git a/pkg/turbine/repair_followup_test.go b/pkg/turbine/repair_followup_test.go new file mode 100644 index 000000000..a8bb6a21f --- /dev/null +++ b/pkg/turbine/repair_followup_test.go @@ -0,0 +1,222 @@ +package turbine + +import ( + "context" + "fmt" + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/gossip" + "github.com/stretchr/testify/require" + "net" + "sync" + "testing" + "time" +) + +// Existing protocol/pacing fixtures model cold admission without packet parsing. +// Production invokes followups only after the full receiver admission path. +func observeRepairForTest(c *repairClient, conn *net.UDPConn, packet []byte, from *net.UDPAddr, sh *Shred) bool { + matched, highest := c.matchShredResponse(packet, from, sh) + if highest { + a := NewSlotAssembler() + s := newRepairSelectionSlot(sh.Slot) + s.shreds[sh.Index] = sh + s.haveLast, s.lastIndex = sh.LastInSlot(), sh.Index + a.slots[sh.Slot] = s + c.followupHighestResponse(conn, a, sh.Slot) + } + return matched +} + +func TestHighestRepairFollowupSelection(t *testing.T) { + a := NewSlotAssembler() + s := newRepairSelectionSlot(50) + a.slots[50] = s + // A recovered first span must not be requested again; the next span + // lacks twelve data but has eight coding, so only four repairs are needed. + addCodedSet(s, 0, 32, 32, seq(0, 31), 0) + addCodedSet(s, 32, 32, 32, seq(32, 51), 8) + s.haveLast, s.lastIndex = true, 63 + r, ok := a.highestRepairFollowup(50) + require.True(t, ok) + require.Equal(t, []uint32{52, 53, 54, 55}, r.MissingDataShreds) + require.False(t, r.NeedHighestDataShred) + for i := uint32(52); i <= 63; i++ { + s.shreds[i] = &Shred{Index: i} + } + _, ok = a.highestRepairFollowup(50) + require.False(t, ok) + delete(s.shreds, 55) + s.completing = true + _, ok = a.highestRepairFollowup(50) + require.False(t, ok) + s.completing = false + a.completedSlots[50] = struct{}{} + _, ok = a.highestRepairFollowup(50) + require.False(t, ok) + delete(a.completedSlots, 50) + a.ResetSlot(50) + _, ok = a.highestRepairFollowup(50) + require.False(t, ok) +} + +func TestHighestRepairFollowupColdAndDiscovery(t *testing.T) { + a := NewSlotAssembler() + s := newRepairSelectionSlot(50) + a.slots[50] = s + s.shreds[600] = &Shred{Index: 600} + r, ok := a.highestRepairFollowup(50) + require.True(t, ok) + require.Len(t, r.MissingDataShreds, 256) + require.Equal(t, uint32(0), r.MissingDataShreds[0]) + require.True(t, r.NeedHighestDataShred) + require.Equal(t, uint32(601), r.HighestDataShredIndex) + // Possession holes only, even without a coding layout. + for i := uint32(0); i < 600; i++ { + s.shreds[i] = &Shred{Index: i} + } + delete(s.shreds, 299) + r, ok = a.highestRepairFollowup(50) + require.True(t, ok) + require.Equal(t, []uint32{299}, r.MissingDataShreds) +} + +func TestHighestRepairFollowupSendsOnlyDeficit(t *testing.T) { + conn, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer conn.Close() + sink, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer sink.Close() + c := newPacingTestClient(t) + c.peerCache = []gossip.RepairPeer{{Addr: sink.LocalAddr().(*net.UDPAddr)}} + c.peerCacheAt = time.Now() + a := NewSlotAssembler() + s := newRepairSelectionSlot(50) + a.slots[50] = s + addCodedSet(s, 0, 32, 32, seq(0, 19), 8) + s.haveLast = true + s.lastIndex = 31 + c.followupHighestResponse(conn, a, 50) + require.Equal(t, uint64(4), c.requests.Load()) + c.followupHighestResponse(conn, a, 50) + require.Equal(t, uint64(4), c.requests.Load()) // same inflight dedupe + // Race a reset with read-only selection; no old slot is recreated. + var wg sync.WaitGroup + wg.Add(1) + go func() { + defer wg.Done() + for i := 0; i < 100; i++ { + a.highestRepairFollowup(50) + } + }() + a.ResetSlot(50) + wg.Wait() + c.followupHighestResponse(conn, a, 50) + require.Equal(t, uint64(4), c.requests.Load()) +} + +func TestReceiverHighestFollowupUsesAdmittedState(t *testing.T) { + packets := fixtures.DataShreds(t, "mainnet", 102815960) + require.Greater(t, len(packets), 12) + conn, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer conn.Close() + sink, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer sink.Close() + r := NewUDPReceiver("127.0.0.1:0") + c := newPacingTestClient(t) + r.repairClient = c + c.peerCache = []gossip.RepairPeer{{Addr: sink.LocalAddr().(*net.UDPAddr)}} + c.peerCacheAt = time.Now() + for _, p := range packets[:11] { + require.True(t, r.processPacket(context.Background(), nil, p, nil, false)) + } + from := &net.UDPAddr{IP: net.IPv4(10, 0, 0, 9), Port: 8009} + addr, _ := repairAddressKeyFromUDP(from) + key := repairRequestKey{kind: repairRequestHighestWindowIndex, slot: 102815960, index: 0} + c.outstanding[key] = outstandingRepairRequest{key: key, nonce: 42, addr: addr, sentAt: time.Now()} + c.byResponse[repairResponseKey{addr: addr, nonce: 42}] = key + packet := append(append([]byte(nil), packets[11]...), nonceTrailer(42)...) + require.True(t, r.processPacket(context.Background(), conn, packet, from, true)) + require.Equal(t, uint64(1), c.requests.Load(), "only continued discovery, no already held indices") + for key := range c.outstanding { + require.Equal(t, repairRequestHighestWindowIndex, key.kind) + require.Equal(t, uint32(12), key.index) + } +} + +// Disk-only discovery must remain repairable after hydration enters the slot. +func TestHighestRepairFollowupAfterDiskOnlyHydration(t *testing.T) { + const slot = uint64(102815960) + packets := fixtures.DataShreds(t, "mainnet", slot) + r := NewUDPReceiver("127.0.0.1:0") + spool, err := OpenShredSpool(t.TempDir(), 0) + require.NoError(t, err) + defer spool.Close() + r.SetShredSpool(spool) + r.assembler.maxObservedSlot = slot + 1000 + r.SetHydrationWindow(slot-8, slot-1) + require.True(t, r.skipAssemblyForSpool(slot)) + // Seed only a partial range on disk; an authentic highest response extends it. + for _, p := range packets[:10] { + require.True(t, r.processPacket(context.Background(), nil, p, nil, false)) + } + c := newPacingTestClient(t) + r.repairClient = c + from := &net.UDPAddr{IP: net.IPv4(10, 0, 0, 9), Port: 8009} + addr, _ := repairAddressKeyFromUDP(from) + key := repairRequestKey{kind: repairRequestHighestWindowIndex, slot: slot, index: 0} + c.outstanding[key] = outstandingRepairRequest{key: key, nonce: 43, addr: addr, sentAt: time.Now()} + c.byResponse[repairResponseKey{addr: addr, nonce: 43}] = key + packet := append(append([]byte(nil), packets[11]...), nonceTrailer(43)...) + require.True(t, r.processPacket(context.Background(), nil, packet, from, true)) + require.Equal(t, uint64(0), c.requests.Load()) + _, live := r.assembler.HeadShredDetail(slot) + require.False(t, live) + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { defer close(done); r.hydrateLoop(ctx) }() + defer func() { cancel(); <-done }() + r.SetRetentionFloor(slot) // replay protects the hydration window during catchup + r.assembler.PrioritizeRepairSlot(slot) + r.SetHydrationWindow(slot, slot) + require.Eventually(t, func() bool { return r.hydratedSlots.Load() > 0 }, time.Second, time.Millisecond) + req, ok := r.assembler.highestRepairFollowup(slot) + require.True(t, ok) + require.Equal(t, []uint32{10}, req.MissingDataShreds) + require.True(t, req.NeedHighestDataShred) + require.Equal(t, uint32(12), req.HighestDataShredIndex) + // Replay priority makes the hole eligible for the ordinary scheduler. + r.assembler.PrioritizeRepairSlot(slot) + priority, _ := r.assembler.RepairRequestsTiered(64, 2048) + require.NotEmpty(t, priority) + require.Equal(t, slot, priority[0].Slot) + require.Equal(t, []uint32{10}, priority[0].MissingDataShreds) +} + +func BenchmarkHighestRepairFollowup(b *testing.B) { + for _, n := range []int{16384, 65536} { + for _, fragmented := range []bool{false, true} { + b.Run(fmt.Sprintf("shreds%d/fragmented%t", n, fragmented), func(b *testing.B) { + a := NewSlotAssembler() + s := newRepairSelectionSlot(50) + a.slots[50] = s + for start := 0; start < n; start += 32 { + last := start + 31 + coding := 0 + if fragmented { + last = start + 19 + coding = 8 + } + addCodedSet(s, uint32(start), 32, 32, seq(uint32(start), uint32(last)), coding) + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + a.highestRepairFollowup(50) + } + }) + } + } +} diff --git a/pkg/turbine/repair_pacing_test.go b/pkg/turbine/repair_pacing_test.go index 8391915a5..ad747cb1a 100644 --- a/pkg/turbine/repair_pacing_test.go +++ b/pkg/turbine/repair_pacing_test.go @@ -90,7 +90,7 @@ func TestLateResponseMatchedAfterExpiry(t *testing.T) { packet := nonceTrailer(777) shred := &Shred{Slot: 42, Index: 3, Type: ShredTypeData} - if !c.observeShredResponse(nil, packet, from, shred) { + if !observeRepairForTest(c, nil, packet, from, shred) { t.Fatal("late answer must be attributed as a repair delivery") } if c.lateResponses.Load() != 1 { @@ -104,7 +104,7 @@ func TestLateResponseMatchedAfterExpiry(t *testing.T) { } // Second delivery of the same nonce: entry consumed, ordinary broadcast. - if c.observeShredResponse(nil, packet, from, shred) { + if observeRepairForTest(c, nil, packet, from, shred) { t.Fatal("expired entry must be single-use") } } @@ -122,7 +122,7 @@ func TestLateResponseWrongSlotRejected(t *testing.T) { c.byResponse[repairResponseKey{addr: addrKey, nonce: 900}] = reqKey c.expireOutstanding(time.Now()) - if c.observeShredResponse(nil, nonceTrailer(900), from, &Shred{Slot: 43, Index: 3, Type: ShredTypeData}) { + if observeRepairForTest(c, nil, nonceTrailer(900), from, &Shred{Slot: 43, Index: 3, Type: ShredTypeData}) { t.Fatal("wrong-slot late answer must not be attributed as repair") } if c.lateResponses.Load() != 0 { @@ -165,7 +165,7 @@ func TestNonConformingResponseRejected(t *testing.T) { c.addInflightLocked(tc.key.shred(), time.Now()) c.mu.Unlock() - if c.observeShredResponse(nil, nonceTrailer(111), from, tc.shred) { + if observeRepairForTest(c, nil, nonceTrailer(111), from, tc.shred) { t.Fatal("non-conforming answer must not be attributed as a repair delivery") } if c.responses.Load() != 0 || c.lateResponses.Load() != 0 { @@ -211,7 +211,7 @@ func TestLateHighestResponseFiresFollowups(t *testing.T) { c.byResponse[repairResponseKey{addr: addrKey, nonce: 6}] = reqKey c.expireOutstanding(time.Now()) - if !c.observeShredResponse(conn, nonceTrailer(6), from, &Shred{Slot: 50, Index: 200, Type: ShredTypeData}) { + if !observeRepairForTest(c, conn, nonceTrailer(6), from, &Shred{Slot: 50, Index: 200, Type: ShredTypeData}) { t.Fatal("late HWI answer must match") } if c.lateResponses.Load() != 1 || c.responses.Load() != 0 { @@ -448,7 +448,7 @@ func TestRepairAnswerCancelsSiblingAttempts(t *testing.T) { // The ORIGINAL peer answers timely; the sibling is neutral-cancelled. from := &net.UDPAddr{IP: sinkAddr.IP, Port: sinkAddr.Port} - if !c.observeShredResponse(conn, nonceTrailer(o0.nonce), from, &Shred{Slot: 60, Index: 3, Type: ShredTypeData}) { + if !observeRepairForTest(c, conn, nonceTrailer(o0.nonce), from, &Shred{Slot: 60, Index: 3, Type: ShredTypeData}) { t.Fatal("original attempt's answer must match") } c.mu.Lock() @@ -568,7 +568,7 @@ func TestFollowupsAreMeteredByTokenBucket(t *testing.T) { // by the primed outstanding entry, leaving a stray token. c.takeRateTokens(repairMaxRequestsPerSecond) c.takeRateTokens(repairMaxRequestsPerSecond) - if !c.observeShredResponse(conn, packet, from, shred) { + if !observeRepairForTest(c, conn, packet, from, shred) { t.Fatal("response itself must match") } if got := c.requests.Load(); got != 0 { @@ -580,7 +580,7 @@ func TestFollowupsAreMeteredByTokenBucket(t *testing.T) { c.rateRefillAt = time.Now() c.rateTokens = repairMaxRequestsPerSecond c.mu.Unlock() - if !c.observeShredResponse(conn, packet, from, shred) { + if !observeRepairForTest(c, conn, packet, from, shred) { t.Fatal("response itself must match") } // A full bucket sends the whole revealed gap: under the adaptive per-peer diff --git a/pkg/turbine/repair_selection_test.go b/pkg/turbine/repair_selection_test.go index 76ebf157f..39cf4985a 100644 --- a/pkg/turbine/repair_selection_test.go +++ b/pkg/turbine/repair_selection_test.go @@ -251,3 +251,77 @@ func TestHeadPolicy(t *testing.T) { t.Fatalf("bulk policy = %+v, want no concurrent duplicate attempts", bulk) } } + +func TestRepairSelectionPrefixBeforeCheapest(t *testing.T) { + s := newRepairSelectionSlot(9) + addCodedSet(s, 0, 32, 32, seq(0, 9), 12) + addCodedSet(s, 32, 32, 32, seq(32, 56), 6) + s.haveLast = true + s.lastIndex = 63 + got := s.missingDataForRepairWithPrefix(63, 4, true) + if !reflect.DeepEqual(got, seq(10, 13)) { + t.Fatalf("prefix %v", got) + } + all := s.missingDataForRepairWithPrefix(63, 256, true) + if !reflect.DeepEqual(all, append(seq(10, 19), 57)) { + t.Fatalf("remaining order %v", all) + } +} + +func TestRepairSelectionUncodedPrefixBeforeCoded(t *testing.T) { + s := newRepairSelectionSlot(9) + addUncodedData(s, seq(1, 31)...) + addCodedSet(s, 32, 32, 32, seq(32, 56), 6) + s.haveLast = true + s.lastIndex = 63 + got := s.missingDataForRepairWithPrefix(63, 256, true) + if !reflect.DeepEqual(got, []uint32{0, 57}) { + t.Fatalf("uncoded prefix %v", got) + } + if got := s.missingDataForRepairWithPrefix(63, 0, true); len(got) != 0 { + t.Fatalf("zero budget: %v", got) + } +} + +func TestRepairSelectionPrefixOnlyForStreamingPriorityHead(t *testing.T) { + a := NewSlotAssembler() + for _, slot := range []uint64{9, 10} { + s := newRepairSelectionSlot(slot) + addCodedSet(s, 0, 32, 32, seq(0, 9), 12) + addCodedSet(s, 32, 32, 32, seq(32, 56), 6) + s.haveLast = true + s.lastIndex = 63 + a.slots[slot] = s + } + a.maxObservedSlot = 11 + a.PrioritizeRepairRange(9, 10) + p, _ := a.RepairRequestsTiered(2, 256) + if len(p) != 2 || p[0].MissingDataShreds[0] != 57 { + t.Fatalf("nonstreaming order %v", p) + } + a.SubscribeStream(make(chan StreamEvent, 1)) + p, _ = a.RepairRequestsTiered(2, 256) + if len(p) != 2 || p[0].MissingDataShreds[0] != 10 || p[1].MissingDataShreds[0] != 57 { + t.Fatalf("streaming head scope %v", p) + } + a.SubscribeStream(nil) + p, _ = a.RepairRequestsTiered(2, 256) + if p[0].MissingDataShreds[0] != 57 { + t.Fatal("disabled streaming retained prefix policy") + } +} + +func TestRepairPriorityParentPinnedAfterChild(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 101 + a.retentionFloor = 100 + a.PrioritizeRepairRange(101, 101) + a.PrioritizeRepairRange(100, 100) + priority, _ := a.RepairRequestsTiered(2, 16) + if len(priority) != 2 || priority[0].Slot != 100 || priority[1].Slot != 101 { + t.Fatalf("parent must precede earlier-pinned child: %+v", priority) + } + if a.priorityRepairOrder[0] != 101 { + t.Fatal("selection changed pin retention order") + } +} diff --git a/pkg/turbine/repair_wakeup_test.go b/pkg/turbine/repair_wakeup_test.go new file mode 100644 index 000000000..d9669534a --- /dev/null +++ b/pkg/turbine/repair_wakeup_test.go @@ -0,0 +1,111 @@ +package turbine + +import ( + "context" + "testing" + "time" +) + +func TestPriorityRepairWakeOnlyForNewPins(t *testing.T) { + r := NewUDPReceiver("127.0.0.1:0") + r.repairClient = &repairClient{priorityWake: make(chan struct{}, 1)} + r.PrioritizeRepairSlot(10) + if len(r.repairClient.priorityWake) != 1 { + t.Fatal("new pin did not wake repair") + } + <-r.repairClient.priorityWake + for i := 0; i < 100; i++ { + r.PrioritizeRepairSlot(10) + } + if len(r.repairClient.priorityWake) != 0 { + t.Fatal("duplicate pins caused wakeups") + } + r.PrioritizeRepairRange(10, 12) + r.PrioritizeRepairSlot(13) + if len(r.repairClient.priorityWake) != 1 { + t.Fatal("new pins should coalesce") + } + <-r.repairClient.priorityWake + r.assembler.completedSlots[14] = struct{}{} + r.PrioritizeRepairSlot(14) + r.PrioritizeRepairSlot(0) + if len(r.repairClient.priorityWake) != 0 { + t.Fatal("completed/invalid slot woke repair") + } +} + +func TestRepairScheduleWakeCoalescingAndCancellation(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + wake := make(chan struct{}, 1) + calls := make(chan time.Time, 4) + entered := make(chan struct{}) + release := make(chan struct{}) + done := make(chan struct{}) + go func() { + defer close(done) + first := true + runRepairSchedule(ctx, wake, time.Hour, 20*time.Millisecond, func() { + calls <- time.Now() + if first { + first = false + close(entered) + <-release + } + }) + }() + wake <- struct{}{} + select { + case <-entered: + case <-time.After(time.Second): + t.Fatal("wake did not bypass periodic timer") + } + for i := 0; i < 100; i++ { + select { + case wake <- struct{}{}: + default: + } + } + first := <-calls + close(release) + select { + case second := <-calls: + if second.Sub(first) < 20*time.Millisecond { + t.Fatal("unbounded scan frequency") + } + case <-time.After(time.Second): + t.Fatal("pending wake lost") + } + cancel() + select { + case <-done: + case <-time.After(time.Second): + t.Fatal("scheduler did not stop") + } + if len(calls) != 0 { + t.Fatal("wake burst caused extra scans") + } +} + +func TestRepairSchedulePeriodicWithoutWake(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + calls := make(chan struct{}, 1) + done := make(chan struct{}) + go func() { + defer close(done) + runRepairSchedule(ctx, nil, time.Millisecond, time.Millisecond, func() { + select { + case calls <- struct{}{}: + default: + } + }) + }() + select { + case <-calls: + case <-time.After(time.Second): + t.Fatal("periodic repair stopped") + } + cancel() + <-done +} diff --git a/pkg/turbine/repairsim/ledger.go b/pkg/turbine/repairsim/ledger.go new file mode 100644 index 000000000..3510fcd6f --- /dev/null +++ b/pkg/turbine/repairsim/ledger.go @@ -0,0 +1,257 @@ +// Package repairsim provides a deterministic, single-process repair harness +// around Mithril's production Turbine shred generator and slot assembler. +// +// The synthetic ledger and network are test infrastructure. Shred parsing, +// Merkle/signature validation, repair selection, Reed-Solomon reconstruction, +// component decoding, transaction verification, and completion accounting are +// production code paths. +package repairsim + +import ( + "crypto/ed25519" + "crypto/sha256" + "fmt" + + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" +) + +const ( + dataShredsPerFEC = 32 + codeShredsPerFEC = 32 +) + +// LedgerConfig controls deterministic canonical-ledger generation. +type LedgerConfig struct { + StartSlot uint64 `json:"start_slot"` + Slots int `json:"slots"` + FECsPerSlot int `json:"fec_sets_per_slot"` + EntriesPerSlot int `json:"entries_per_slot,omitempty"` + Seed int64 `json:"seed"` + ShredVersion uint16 `json:"shred_version"` + ReferenceTick uint8 `json:"reference_tick"` +} + +// Packet is one canonical wire packet and its parsed routing metadata. +type Packet struct { + Bytes []byte + Slot uint64 + Type turbine.ShredType + Index uint32 + FECSetIndex uint32 + Position uint16 +} + +// FECSet contains the canonical packets for one 32+32 FEC set. +type FECSet struct { + Index uint32 + Data []Packet + Coding []Packet +} + +// Slot is the complete canonical source for one generated slot. +type Slot struct { + Number uint64 + ParentSlot uint64 + Entries []turbine.Entry + FECs []FECSet + Data map[uint32]Packet + Highest uint32 +} + +// Ledger is the complete data held by the in-process repair peer. +type Ledger struct { + Config LedgerConfig + Leader solana.PrivateKey + LeaderPub solana.PublicKey + Slots []Slot + bySlot map[uint64]*Slot +} + +// Slot returns a canonical slot by number. +func (l *Ledger) Slot(number uint64) (*Slot, bool) { + if l == nil { + return nil, false + } + s, ok := l.bySlot[number] + return s, ok +} + +// GenerateLedger creates authentic signed Merkle shreds using the production +// 32+32 generator. Entries contain no transactions: this isolates repair, +// reconstruction, parsing, and storage while still exercising the real block +// component codec and completion path. +func GenerateLedger(cfg LedgerConfig) (*Ledger, error) { + if cfg.Slots <= 0 { + return nil, fmt.Errorf("slots must be positive") + } + if cfg.FECsPerSlot <= 0 { + return nil, fmt.Errorf("FEC sets per slot must be positive") + } + if cfg.StartSlot == 0 { + cfg.StartSlot = 10_000 + } + if cfg.ReferenceTick == 0 { + cfg.ReferenceTick = 63 + } + + leader := deterministicLeader(cfg.Seed) + entryCount := cfg.EntriesPerSlot + if entryCount == 0 { + var err error + entryCount, err = findEntryCount(cfg, leader) + if err != nil { + return nil, err + } + } + + ledger := &Ledger{ + Config: cfg, + Leader: leader, + LeaderPub: leader.PublicKey(), + Slots: make([]Slot, 0, cfg.Slots), + bySlot: make(map[uint64]*Slot, cfg.Slots), + } + ledger.Config.EntriesPerSlot = entryCount + for i := 0; i < cfg.Slots; i++ { + number := cfg.StartSlot + uint64(i) + parent := number - 1 + entries := deterministicEntries(cfg.Seed, number, entryCount) + slot, err := generateSlot(cfg, leader, number, parent, entries) + if err != nil { + return nil, fmt.Errorf("generate slot %d: %w", number, err) + } + if got := len(slot.FECs); got != cfg.FECsPerSlot { + return nil, fmt.Errorf("slot %d produced %d FEC sets, want %d (entries=%d)", number, got, cfg.FECsPerSlot, entryCount) + } + ledger.Slots = append(ledger.Slots, slot) + ledger.bySlot[number] = &ledger.Slots[len(ledger.Slots)-1] + } + return ledger, nil +} + +func deterministicLeader(seed int64) solana.PrivateKey { + var input [16]byte + for i := range input { + input[i] = byte(uint64(seed)>>uint((i%8)*8)) ^ byte(i*29+7) + } + digest := sha256.Sum256(append([]byte("mithril-repair-sim-leader-v1"), input[:]...)) + return solana.PrivateKey(ed25519.NewKeyFromSeed(digest[:])) +} + +func deterministicEntries(seed int64, slot uint64, count int) []turbine.Entry { + entries := make([]turbine.Entry, count) + for i := range entries { + material := fmt.Sprintf("mithril-repair-sim-entry-v1:%d:%d:%d", seed, slot, i) + h := sha256.Sum256([]byte(material)) + entries[i] = turbine.Entry{NumHashes: 1, Hash: solana.Hash(h)} + } + return entries +} + +// findEntryCount uses the generator itself as the capacity oracle. This avoids +// duplicating signed-last-FEC payload constants in the harness. +func findEntryCount(cfg LedgerConfig, leader solana.PrivateKey) (int, error) { + lo, hi := 1, cfg.FECsPerSlot*900 + for lo < hi { + mid := lo + (hi-lo)/2 + entries := deterministicEntries(cfg.Seed, cfg.StartSlot, mid) + slot, err := generateSlot(cfg, leader, cfg.StartSlot, cfg.StartSlot-1, entries) + if err != nil { + return 0, err + } + if len(slot.FECs) < cfg.FECsPerSlot { + lo = mid + 1 + } else { + hi = mid + } + } + entries := deterministicEntries(cfg.Seed, cfg.StartSlot, lo) + slot, err := generateSlot(cfg, leader, cfg.StartSlot, cfg.StartSlot-1, entries) + if err != nil { + return 0, err + } + if len(slot.FECs) != cfg.FECsPerSlot { + return 0, fmt.Errorf("cannot derive %d FEC sets within %d entries (got %d)", cfg.FECsPerSlot, hi, len(slot.FECs)) + } + return lo, nil +} + +func generateSlot(cfg LedgerConfig, leader solana.PrivateKey, number, parent uint64, entries []turbine.Entry) (Slot, error) { + component, err := turbine.NewEntryBatch(entries) + if err != nil { + return Slot{}, err + } + shredder := turbine.Shredder{ + Slot: number, + ParentSlot: parent, + Version: cfg.ShredVersion, + ReferenceTick: cfg.ReferenceTick, + } + batch, _, _, err := shredder.MakeMerkleShredsFromComponent( + leader, component, true, solana.Hash{}, 0, 0, + ) + if err != nil { + return Slot{}, err + } + + byFEC := make(map[uint32]*FECSet) + packetByKey := make(map[packetKey][]byte, len(batch.Packets)) + for _, raw := range batch.Packets { + shred, err := turbine.ParseShred(raw) + if err != nil { + return Slot{}, err + } + key := keyForShred(shred) + packetByKey[key] = append([]byte(nil), raw...) + } + data := make(map[uint32]Packet, len(batch.DataShreds)) + var highest uint32 + for _, shred := range append(append([]*turbine.Shred(nil), batch.DataShreds...), batch.CodeShreds...) { + fec := byFEC[shred.FECSetIndex] + if fec == nil { + fec = &FECSet{Index: shred.FECSetIndex} + byFEC[shred.FECSetIndex] = fec + } + packet := Packet{ + Bytes: packetByKey[keyForShred(shred)], + Slot: shred.Slot, + Type: shred.Type, + Index: shred.Index, + FECSetIndex: shred.FECSetIndex, + Position: shred.Position, + } + if shred.Type == turbine.ShredTypeData { + fec.Data = append(fec.Data, packet) + data[shred.Index] = packet + if shred.Index > highest { + highest = shred.Index + } + } else { + fec.Coding = append(fec.Coding, packet) + } + } + fecs := make([]FECSet, 0, len(byFEC)) + for index := uint32(0); len(fecs) < len(byFEC); index += dataShredsPerFEC { + fec := byFEC[index] + if fec == nil { + return Slot{}, fmt.Errorf("non-contiguous FEC sets: missing %d", index) + } + if len(fec.Data) != dataShredsPerFEC || len(fec.Coding) != codeShredsPerFEC { + return Slot{}, fmt.Errorf("FEC %d has %d+%d shreds", index, len(fec.Data), len(fec.Coding)) + } + fecs = append(fecs, *fec) + } + return Slot{Number: number, ParentSlot: parent, Entries: entries, FECs: fecs, Data: data, Highest: highest}, nil +} + +type packetKey struct { + type_ turbine.ShredType + index uint32 + fec uint32 + position uint16 +} + +func keyForShred(shred *turbine.Shred) packetKey { + return packetKey{type_: shred.Type, index: shred.Index, fec: shred.FECSetIndex, position: shred.Position} +} diff --git a/pkg/turbine/repairsim/prefix_repair_test.go b/pkg/turbine/repairsim/prefix_repair_test.go new file mode 100644 index 000000000..9077b0649 --- /dev/null +++ b/pkg/turbine/repairsim/prefix_repair_test.go @@ -0,0 +1,107 @@ +package repairsim + +import ( + "fmt" + "reflect" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/turbine" +) + +// Logical-time experiment using production selection, authenticated packets and +// FEC recovery. A fixed request budget per 20ms round; this measures when the +// first data span is restored, not execution latency or production retry policy. +func TestStreamingPrefixRepairUnderLimitedBudget(t *testing.T) { + ledger := testLedger(t, 1, 8) + for _, budget := range []int{1, 2, 4, 16} { + t.Run(fmt.Sprint(budget), func(t *testing.T) { + type result struct { + prefix, complete time.Duration + requests int + block *block.Block + } + run := func(stream bool) result { + a := turbine.NewSlotAssembler() + if stream { + a.SubscribeStream(make(chan turbine.StreamEvent, 256)) + } + slot := ledger.Slots[0] + feed := func(p Packet) *block.Block { + sh, err := parseAndVerify(p, ledger) + if err != nil { + t.Fatal(err) + } + b, err := a.AddShred(sh) + if err != nil { + t.Fatal(err) + } + return b + } + for j, f := range slot.FECs { + for _, p := range f.Data[:29] { + feed(p) + } + n := 2 + if j == 0 { + n = 1 + } + for _, p := range f.Coding[:n] { + feed(p) + } + } + a.PrioritizeRepairSlot(slot.Number) + r := result{} + for round := 1; round <= 20; round++ { + req := a.RepairRequests(1, 256) + if len(req) == 0 { + t.Fatal("unfinished slot has no repair request") + } + indices := req[0].MissingDataShreds + if len(indices) > budget { + indices = indices[:budget] + } + if len(indices) == 0 { + t.Fatal("no exact repair work") + } + for _, idx := range indices { + r.requests++ + if b := feed(slot.Data[idx]); b != nil { + r.block = b + r.complete = time.Duration(round) * 20 * time.Millisecond + } + } + remaining := a.RepairRequests(1, 256) + hole := false + for _, q := range remaining { + for _, idx := range q.MissingDataShreds { + if idx < 32 { + hole = true + } + } + } + if !hole && r.prefix == 0 { + r.prefix = time.Duration(round) * 20 * time.Millisecond + } + if r.block != nil { + return r + } + } + t.Fatal("did not complete") + return r + } + before, after := run(false), run(true) + t.Logf("first span %v -> %v; full block %v -> %v; requests %d -> %d", before.prefix, after.prefix, before.complete, after.complete, before.requests, after.requests) + if after.prefix > before.prefix || (budget < 9 && after.prefix == before.prefix) { + t.Fatal("prefix did not improve") + } + if after.requests != before.requests || after.complete > before.complete { + t.Fatal("request count or completion regressed") + } + if !reflect.DeepEqual(before.block.Transactions, after.block.Transactions) || !reflect.DeepEqual(before.block.Entries, after.block.Entries) { + t.Fatal("assembled payload mismatch") + } + }) + } +} diff --git a/pkg/turbine/repairsim/sim.go b/pkg/turbine/repairsim/sim.go new file mode 100644 index 000000000..c89e867e1 --- /dev/null +++ b/pkg/turbine/repairsim/sim.go @@ -0,0 +1,713 @@ +package repairsim + +import ( + "container/heap" + "errors" + "fmt" + "math/rand" + "os" + "runtime" + "sort" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/turbine" +) + +type Scenario string + +const ( + ScenarioNearTip Scenario = "near-tip" + ScenarioDeepCatchup Scenario = "deep-catchup" +) + +type Availability string + +const ( + AvailabilityComplete Availability = "complete" + AvailabilityNearLoss Availability = "near-loss" + AvailabilitySparse Availability = "sparse" + AvailabilityMixed Availability = "mixed" +) + +// Config controls the deterministic network and local starting state. +type Config struct { + Scenario Scenario `json:"scenario"` + Availability Availability `json:"availability"` + RepairEnabled bool `json:"repair_enabled"` + RepairLatency time.Duration `json:"repair_latency_ns"` + RepairJitter time.Duration `json:"repair_jitter_ns"` + PacketLoss float64 `json:"packet_loss"` + DuplicateProbability float64 `json:"duplicate_probability"` + BandwidthBytesPerSec int64 `json:"bandwidth_bytes_per_sec"` + MaxConcurrent int `json:"max_concurrent_requests"` + MaxRequestSlots int `json:"max_request_slots"` + MaxMissingPerSlot int `json:"max_missing_per_slot"` + CorruptResponses int `json:"corrupt_responses"` + NaturalLateShreds bool `json:"natural_late_shreds"` + CollectTrace bool `json:"collect_trace"` + Seed int64 `json:"seed"` + SpoolDir string `json:"spool_dir,omitempty"` + SpoolMaxBytes int64 `json:"spool_max_bytes"` +} + +// DefaultConfig returns a deterministic starting point for a scenario. +func DefaultConfig(scenario Scenario) Config { + cfg := Config{ + Scenario: scenario, + RepairEnabled: true, + RepairLatency: 20 * time.Millisecond, + RepairJitter: 2 * time.Millisecond, + DuplicateProbability: 0.02, + BandwidthBytesPerSec: 100 * 1024 * 1024, + MaxConcurrent: 256, + MaxRequestSlots: 64, + MaxMissingPerSlot: 256, + NaturalLateShreds: scenario == ScenarioNearTip, + CollectTrace: true, + Seed: 1, + SpoolMaxBytes: 1 << 30, + } + if scenario == ScenarioDeepCatchup { + cfg.Availability = AvailabilityMixed + } else { + cfg.Availability = AvailabilityNearLoss + } + return cfg +} + +// TraceEvent is a deterministic logical-time event. CPU durations are kept in +// Result.StageCPU so trace equality does not depend on scheduler noise. +type TraceEvent struct { + Sequence int `json:"sequence"` + AtNanos int64 `json:"at_ns"` + Stage string `json:"stage"` + Slot uint64 `json:"slot,omitempty"` + FECSetIndex uint32 `json:"fec_set_index,omitempty"` + ShredIndex uint32 `json:"shred_index,omitempty"` + ShredType turbine.ShredType `json:"shred_type,omitempty"` + Bytes int `json:"bytes,omitempty"` + Detail string `json:"detail,omitempty"` +} + +type LatencySummary struct { + P50 time.Duration `json:"p50_ns"` + P95 time.Duration `json:"p95_ns"` + P99 time.Duration `json:"p99_ns"` +} + +// Result separates simulated-network time from actual local execution time. +type Result struct { + Scenario Scenario `json:"scenario"` + Availability Availability `json:"availability"` + Slots int `json:"slots"` + CompletedSlots int `json:"completed_slots"` + LogicalElapsed time.Duration `json:"logical_elapsed_ns"` + WallElapsed time.Duration `json:"wall_elapsed_ns"` + TimeToFirstReplayable time.Duration `json:"time_to_first_replayable_ns"` + TimeToFirstRecoveredData time.Duration `json:"time_to_first_recovered_data_ns"` + RepairEligibleToRecovery time.Duration `json:"repair_eligible_to_first_recovery_ns"` + CompletionLatency LatencySummary `json:"completion_latency"` + SlotsPerLogicalSecond float64 `json:"slots_per_logical_second"` + SlotsPerCPUSecond float64 `json:"slots_per_cpu_second"` + RepairRequests uint64 `json:"repair_requests"` + RepairResponses uint64 `json:"repair_responses"` + RepairBytesRequested uint64 `json:"repair_bytes_requested"` + RepairBytesReceived uint64 `json:"repair_bytes_received"` + UsefulNetworkDataShreds uint64 `json:"useful_network_data_shreds"` + LocallyRecoveredDataShreds uint64 `json:"locally_recovered_data_shreds"` + LocallyRecoveredDataBytes uint64 `json:"locally_recovered_data_bytes"` + InitialMissingDataShreds uint64 `json:"initial_missing_data_shreds"` + FractionRecoveredLocally float64 `json:"fraction_missing_recovered_locally"` + FECDecodes uint64 `json:"fec_decodes"` + DuplicateResponses uint64 `json:"duplicate_responses"` + CanceledOrLateResponses uint64 `json:"canceled_or_late_responses"` + LostResponses uint64 `json:"lost_responses"` + RejectedCorruptResponses uint64 `json:"rejected_corrupt_responses"` + ShredSignatureCacheHits uint64 `json:"shred_signature_cache_hits"` + ShredEd25519Verifications uint64 `json:"shred_ed25519_verifications"` + QueueHighWater int `json:"queue_high_water"` + SpoolBytes int64 `json:"spool_bytes"` + SpoolCompleteSlots int `json:"spool_complete_slots"` + DataShredBytesReplayable uint64 `json:"data_shred_bytes_replayable"` + Allocations uint64 `json:"allocations"` + StageCPU map[string]time.Duration `json:"stage_cpu_ns"` + Trace []TraceEvent `json:"trace"` + Limitations []string `json:"limitations"` +} + +// Run executes the virtual network around production parsing, validation, +// repair selection, reconstruction, spool insertion, and block completion. +func Run(ledger *Ledger, cfg Config) (Result, error) { + if ledger == nil || len(ledger.Slots) == 0 { + return Result{}, errors.New("empty ledger") + } + if cfg.Scenario != ScenarioNearTip && cfg.Scenario != ScenarioDeepCatchup { + return Result{}, fmt.Errorf("unsupported scenario %q", cfg.Scenario) + } + if cfg.Availability == "" { + cfg.Availability = DefaultConfig(cfg.Scenario).Availability + } + if cfg.MaxConcurrent <= 0 { + cfg.MaxConcurrent = 1 + } + if cfg.MaxRequestSlots <= 0 { + cfg.MaxRequestSlots = 64 + } + if cfg.MaxMissingPerSlot <= 0 { + cfg.MaxMissingPerSlot = 256 + } + if cfg.SpoolMaxBytes <= 0 { + cfg.SpoolMaxBytes = 1 << 30 + } + + spoolDir := cfg.SpoolDir + if spoolDir == "" { + var err error + spoolDir, err = os.MkdirTemp("", "mithril-repair-sim-") + if err != nil { + return Result{}, err + } + defer os.RemoveAll(spoolDir) + } + spool, err := turbine.OpenShredSpool(spoolDir, cfg.SpoolMaxBytes) + if err != nil { + return Result{}, err + } + defer spool.Close() + + var memBefore runtime.MemStats + runtime.ReadMemStats(&memBefore) + wallStarted := time.Now() + s := &simulation{ + ledger: ledger, + cfg: cfg, + assembler: turbine.NewSlotAssembler(), + spool: spool, + rng: rand.New(rand.NewSource(cfg.Seed)), + firstShred: make(map[uint64]time.Duration), + completed: make(map[uint64]*block.Block), + replayableAt: make(map[uint64]time.Duration), + pending: make(map[repairKey]struct{}), + stageCPU: make(map[string]time.Duration), + } + s.assembler.SetRetentionFloor(ledger.Slots[0].Number) + s.assembler.SetOnComplete(spool.MarkComplete) + s.nextReplaySlot = ledger.Slots[0].Number + + if err := s.seedLocalState(); err != nil { + return Result{}, err + } + if err := s.runDeliveries(); err != nil { + return Result{}, err + } + + var memAfter runtime.MemStats + runtime.ReadMemStats(&memAfter) + _, spoolBytes := spool.Stats() + result := s.result + result.Scenario = cfg.Scenario + result.Availability = cfg.Availability + result.Slots = len(ledger.Slots) + result.CompletedSlots = len(s.completed) + result.LogicalElapsed = s.now + result.WallElapsed = time.Since(wallStarted) + result.LocallyRecoveredDataShreds = s.assembler.RecoveredDataShreds() + result.InitialMissingDataShreds = initialMissingDataShreds(ledger, cfg.Availability) + if result.InitialMissingDataShreds > 0 { + result.FractionRecoveredLocally = float64(result.LocallyRecoveredDataShreds) / float64(result.InitialMissingDataShreds) + } + if len(ledger.Slots[0].FECs) > 0 && len(ledger.Slots[0].FECs[0].Data) > 0 { + result.LocallyRecoveredDataBytes = result.LocallyRecoveredDataShreds * uint64(len(ledger.Slots[0].FECs[0].Data[0].Bytes)) + } + if s.haveRecoverAt { + result.TimeToFirstRecoveredData = s.firstRecoverAt + if s.haveRepairAt && s.firstRecoverAt >= s.firstRepairAt { + result.RepairEligibleToRecovery = s.firstRecoverAt - s.firstRepairAt + } + } + result.SpoolBytes = spoolBytes + result.SpoolCompleteSlots = spool.CompleteSlots() + result.Allocations = memAfter.Mallocs - memBefore.Mallocs + result.StageCPU = s.stageCPU + result.Trace = s.trace + result.ShredSignatureCacheHits, result.ShredEd25519Verifications = s.shredVerifier.Stats() + result.Limitations = []string{ + "remote peers and latency are simulated in process; no UDP/IP stack is measured", + "synthetic entries contain no transactions, so transaction execution is not measured", + "slot offered to replay means SlotAssembler emitted a verified block; replay execution is not run", + "FEC decode start is observed at the AddShredFrom call boundary, not inside the Reed-Solomon library", + } + latencies := make([]time.Duration, 0, len(s.replayableAt)) + for slot, completedAt := range s.replayableAt { + if first, ok := s.firstShred[slot]; ok { + latencies = append(latencies, completedAt-first) + } + } + result.CompletionLatency = summarizeLatencies(latencies) + if len(s.replayableAt) > 0 { + first := ledger.Slots[0].Number + result.TimeToFirstReplayable = s.replayableAt[first] + } + if result.LogicalElapsed > 0 { + result.SlotsPerLogicalSecond = float64(result.CompletedSlots) / result.LogicalElapsed.Seconds() + } + if result.WallElapsed > 0 { + result.SlotsPerCPUSecond = float64(result.CompletedSlots) / result.WallElapsed.Seconds() + } + return result, nil +} + +type simulation struct { + ledger *Ledger + cfg Config + assembler *turbine.SlotAssembler + shredVerifier turbine.ShredSignatureVerifier + spool *turbine.ShredSpool + rng *rand.Rand + now time.Duration + nextWireAt time.Duration + sequence int + trace []TraceEvent + queue deliveryHeap + pending map[repairKey]struct{} + firstShred map[uint64]time.Duration + completed map[uint64]*block.Block + replayableAt map[uint64]time.Duration + nextReplaySlot uint64 + stageCPU map[string]time.Duration + result Result + corruptLeft int + firstRepairAt time.Duration + haveRepairAt bool + firstRecoverAt time.Duration + haveRecoverAt bool +} + +func (s *simulation) seedLocalState() error { + s.corruptLeft = s.cfg.CorruptResponses + packetsBySlot := make([][]Packet, len(s.ledger.Slots)) + for slotIdx := range s.ledger.Slots { + packetsBySlot[slotIdx] = initialPackets(&s.ledger.Slots[slotIdx], s.cfg.Availability) + } + for ordinal := 0; ; ordinal++ { + added := false + for slotIdx := range s.ledger.Slots { + packets := packetsBySlot[slotIdx] + if ordinal >= len(packets) { + continue + } + added = true + if err := s.ingest(packets[ordinal], false, false); err != nil { + return fmt.Errorf("seed slot %d: %w", s.ledger.Slots[slotIdx].Number, err) + } + s.now++ + } + if !added { + break + } + } + if s.cfg.NaturalLateShreds && s.cfg.Availability == AvailabilityNearLoss { + for i := range s.ledger.Slots { + slot := &s.ledger.Slots[i] + if len(slot.FECs) == 0 || len(slot.FECs[0].Data) < 31 { + continue + } + at := s.now + s.cfg.RepairLatency/2 + time.Duration(i)*time.Microsecond + heap.Push(&s.queue, delivery{at: at, sequence: s.sequence, packet: slot.FECs[0].Data[30]}) + s.sequence++ + } + } + return nil +} + +func initialPackets(slot *Slot, availability Availability) []Packet { + var out []Packet + for i := range slot.FECs { + fec := &slot.FECs[i] + switch availability { + case AvailabilityComplete: + out = append(out, fec.Data...) + case AvailabilityNearLoss: + if i%2 == 0 { + out = append(out, fec.Data[:30]...) + out = append(out, fec.Coding[0]) + } else { + out = append(out, fec.Data[:31]...) + } + case AvailabilitySparse: + out = append(out, fec.Data[:2]...) + case AvailabilityMixed: + out = append(out, fec.Data[:16]...) + out = append(out, fec.Coding[:15]...) + } + } + return out +} + +func (s *simulation) runDeliveries() error { + const maxIterations = 10_000_000 + for iterations := 0; len(s.completed) < len(s.ledger.Slots); iterations++ { + if iterations >= maxIterations { + return errors.New("repair simulation exceeded iteration limit") + } + if s.cfg.RepairEnabled { + s.prioritizeHeadWindow() + s.scheduleRequests() + } + if len(s.queue) == 0 { + // Without repair, exhaust the live arrivals and report any remaining + // holes. No more traffic is expected to complete those slots. + if !s.cfg.RepairEnabled { + return nil + } + return fmt.Errorf("repair stalled with %d/%d completed", len(s.completed), len(s.ledger.Slots)) + } + event := heap.Pop(&s.queue).(delivery) + if event.at > s.now { + s.now = event.at + } + if event.primary { + delete(s.pending, event.key) + } + if event.drop { + s.result.LostResponses++ + s.record("repair_response_lost", event.packet, "") + continue + } + if event.duplicate { + s.result.DuplicateResponses++ + } + if event.fromRepair { + s.result.RepairResponses++ + s.result.RepairBytesReceived += uint64(len(event.packet.Bytes)) + } + beforeUseful := s.assembler.UsefulRepairShreds() + if err := s.ingest(event.packet, event.fromRepair, event.corrupt); err != nil { + if event.corrupt && errors.Is(err, turbine.ErrInvalidSignature) { + s.result.RejectedCorruptResponses++ + s.record("repair_response_rejected", event.packet, "invalid signature or Merkle proof") + continue + } + return err + } + if event.fromRepair { + afterUseful := s.assembler.UsefulRepairShreds() + if afterUseful == beforeUseful { + s.result.CanceledOrLateResponses++ + } else { + s.result.UsefulNetworkDataShreds += afterUseful - beforeUseful + } + } + } + return nil +} + +func (s *simulation) prioritizeHeadWindow() { + var head uint64 + found := false + for i := range s.ledger.Slots { + slot := s.ledger.Slots[i].Number + if _, complete := s.completed[slot]; !complete { + head, found = slot, true + break + } + } + if !found { + return + } + end := head + 63 + last := s.ledger.Slots[len(s.ledger.Slots)-1].Number + if end > last { + end = last + } + s.assembler.PrioritizeRepairRange(head, end) +} + +func (s *simulation) scheduleRequests() { + capacity := s.cfg.MaxConcurrent - len(s.pending) + if capacity <= 0 { + return + } + requests := s.assembler.RepairRequests(s.cfg.MaxRequestSlots, s.cfg.MaxMissingPerSlot) + for _, req := range requests { + if !s.haveRepairAt { + s.firstRepairAt = s.now + s.haveRepairAt = true + } + s.recordAt("repair_needed_decision", req.Slot, 0, 0, 0, 0, + fmt.Sprintf("missing=%d need_highest=%t", len(req.MissingDataShreds), req.NeedHighestDataShred)) + slot, ok := s.ledger.Slot(req.Slot) + if !ok { + continue + } + indexes := append([]uint32(nil), req.MissingDataShreds...) + if req.NeedHighestDataShred { + indexes = append(indexes, slot.Highest) + } + seen := make(map[uint32]struct{}, len(indexes)) + for _, index := range indexes { + if capacity == 0 { + return + } + if _, duplicate := seen[index]; duplicate { + continue + } + seen[index] = struct{}{} + packet, ok := slot.Data[index] + if !ok { + continue + } + key := repairKey{slot: req.Slot, index: index} + if _, outstanding := s.pending[key]; outstanding { + continue + } + s.pending[key] = struct{}{} + capacity-- + s.result.RepairRequests++ + s.result.RepairBytesRequested += uint64(len(packet.Bytes)) + s.record("repair_request_enqueue", packet, "data shred") + s.record("repair_request_send", packet, "data shred") + s.scheduleResponse(key, packet) + } + } +} + +func (s *simulation) scheduleResponse(key repairKey, packet Packet) { + jitter := time.Duration(0) + if s.cfg.RepairJitter > 0 { + span := int64(s.cfg.RepairJitter)*2 + 1 + jitter = time.Duration(s.rng.Int63n(span)) - s.cfg.RepairJitter + } + at := s.now + s.cfg.RepairLatency + jitter + if at < s.now { + at = s.now + } + if at < s.nextWireAt { + at = s.nextWireAt + } + if s.cfg.BandwidthBytesPerSec > 0 { + wire := time.Duration(float64(len(packet.Bytes)) / float64(s.cfg.BandwidthBytesPerSec) * float64(time.Second)) + if wire < time.Nanosecond { + wire = time.Nanosecond + } + at += wire + s.nextWireAt = at + } + d := delivery{at: at, sequence: s.sequence, packet: packet, key: key, primary: true, fromRepair: true} + s.sequence++ + if s.cfg.PacketLoss > 0 && s.rng.Float64() < s.cfg.PacketLoss { + d.drop = true + } + if s.corruptLeft > 0 { + d.corrupt = true + s.corruptLeft-- + } + heap.Push(&s.queue, d) + if !d.drop && s.cfg.DuplicateProbability > 0 && s.rng.Float64() < s.cfg.DuplicateProbability { + dup := d + dup.at++ + dup.sequence = s.sequence + dup.primary = false + dup.duplicate = true + dup.corrupt = false + s.sequence++ + heap.Push(&s.queue, dup) + } + if len(s.queue) > s.result.QueueHighWater { + s.result.QueueHighWater = len(s.queue) + } +} + +func (s *simulation) ingest(packet Packet, fromRepair, corrupt bool) error { + raw := packet.Bytes + if corrupt { + raw = append([]byte(nil), raw...) + if len(raw) > 200 { + raw[200] ^= 0x80 + } else if len(raw) > 0 { + raw[len(raw)-1] ^= 0x80 + } + } + started := time.Now() + shred, err := turbine.ParseShred(raw) + s.stageCPU["shred_parse"] += time.Since(started) + if err != nil { + return err + } + started = time.Now() + err = s.shredVerifier.Verify(shred, s.ledger.LeaderPub) + s.stageCPU["shred_validation"] += time.Since(started) + if err != nil { + return err + } + s.record("shred_validation", packet, "Merkle proof and leader signature valid") + if _, ok := s.firstShred[shred.Slot]; !ok { + s.firstShred[shred.Slot] = s.now + s.record("first_shred", packet, "") + } + started = time.Now() + spooled := s.spool.AppendShred(shred, raw) + s.stageCPU["blockstore_insert"] += time.Since(started) + if spooled { + s.record("blockstore_insert", packet, "verified shred spool") + } + beforeRecovered := s.assembler.RecoveredDataShreds() + started = time.Now() + blk, err := s.assembler.AddShredFrom(shred, fromRepair) + s.stageCPU["assembler_ingest_and_recovery"] += time.Since(started) + if err != nil { + return err + } + afterRecovered := s.assembler.RecoveredDataShreds() + if afterRecovered > beforeRecovered { + s.result.FECDecodes++ + if !s.haveRecoverAt { + s.firstRecoverAt = s.now + s.haveRecoverAt = true + } + s.record("fec_threshold_reached", packet, "observed at AddShredFrom call boundary") + s.record("fec_decode_complete", packet, fmt.Sprintf("recovered_data=%d", afterRecovered-beforeRecovered)) + } + if fromRepair { + s.record("repair_response_receive", packet, "") + } else { + s.record("live_shred_receive", packet, "") + } + if blk != nil { + canonical, ok := s.ledger.Slot(blk.Slot) + if !ok { + return fmt.Errorf("completed unknown slot %d", blk.Slot) + } + if err := compareBlock(canonical, blk); err != nil { + return err + } + if _, ok := s.spool.IsComplete(blk.Slot); !ok { + return fmt.Errorf("slot %d completed without spool completion record", blk.Slot) + } + s.completed[blk.Slot] = blk + for _, data := range canonical.Data { + s.result.DataShredBytesReplayable += uint64(len(data.Bytes)) + } + s.record("slot_offered_to_replay", packet, "verified block emitted") + s.advanceReplayable() + } + return nil +} + +func (s *simulation) advanceReplayable() { + for { + if _, ok := s.completed[s.nextReplaySlot]; !ok { + return + } + s.replayableAt[s.nextReplaySlot] = s.now + s.recordAt("slot_replayable", s.nextReplaySlot, 0, 0, 0, 0, "contiguous parent chain available") + s.nextReplaySlot++ + } +} + +func compareBlock(canonical *Slot, got *block.Block) error { + if got.Slot != canonical.Number || got.SourceParentSlot != canonical.ParentSlot { + return fmt.Errorf("block identity got slot=%d parent=%d, want slot=%d parent=%d", got.Slot, got.SourceParentSlot, canonical.Number, canonical.ParentSlot) + } + if len(got.Entries) != len(canonical.Entries) { + return fmt.Errorf("slot %d entries=%d, want %d", got.Slot, len(got.Entries), len(canonical.Entries)) + } + for i := range canonical.Entries { + want := canonical.Entries[i] + entry := got.Entries[i] + if entry.NumHashes != want.NumHashes || string(entry.Hash) != string(want.Hash[:]) || len(entry.Indices) != len(want.Txns) { + return fmt.Errorf("slot %d entry %d differs from canonical ledger", got.Slot, i) + } + } + if !got.TransactionSignaturesVerified() { + return fmt.Errorf("slot %d block was not signature-verified", got.Slot) + } + return nil +} + +func (s *simulation) record(stage string, packet Packet, detail string) { + s.recordAt(stage, packet.Slot, packet.FECSetIndex, packet.Index, packet.Type, len(packet.Bytes), detail) +} + +func (s *simulation) recordAt(stage string, slot uint64, fec, index uint32, typ turbine.ShredType, bytes int, detail string) { + if s.cfg.CollectTrace { + s.trace = append(s.trace, TraceEvent{ + Sequence: s.sequence, AtNanos: int64(s.now), Stage: stage, Slot: slot, + FECSetIndex: fec, ShredIndex: index, ShredType: typ, Bytes: bytes, Detail: detail, + }) + } + s.sequence++ +} + +func summarizeLatencies(values []time.Duration) LatencySummary { + if len(values) == 0 { + return LatencySummary{} + } + sort.Slice(values, func(i, j int) bool { return values[i] < values[j] }) + percentile := func(p float64) time.Duration { + index := int(float64(len(values)-1)*p + 0.5) + return values[index] + } + return LatencySummary{P50: percentile(.50), P95: percentile(.95), P99: percentile(.99)} +} + +func initialMissingDataShreds(ledger *Ledger, availability Availability) uint64 { + var heldPerFEC int + switch availability { + case AvailabilityComplete: + heldPerFEC = 32 + case AvailabilityNearLoss: + var missing uint64 + for i := range ledger.Slots { + for fec := range ledger.Slots[i].FECs { + if fec%2 == 0 { + missing += 2 + } else { + missing++ + } + } + } + return missing + case AvailabilitySparse: + heldPerFEC = 2 + case AvailabilityMixed: + heldPerFEC = 16 + } + return uint64(len(ledger.Slots) * ledger.Config.FECsPerSlot * (dataShredsPerFEC - heldPerFEC)) +} + +type repairKey struct { + slot uint64 + index uint32 +} + +type delivery struct { + at time.Duration + sequence int + packet Packet + key repairKey + primary bool + fromRepair bool + drop bool + duplicate bool + corrupt bool +} + +type deliveryHeap []delivery + +func (h deliveryHeap) Len() int { return len(h) } +func (h deliveryHeap) Less(i, j int) bool { + if h[i].at != h[j].at { + return h[i].at < h[j].at + } + return h[i].sequence < h[j].sequence +} +func (h deliveryHeap) Swap(i, j int) { h[i], h[j] = h[j], h[i] } +func (h *deliveryHeap) Push(x any) { *h = append(*h, x.(delivery)) } +func (h *deliveryHeap) Pop() any { + old := *h + last := old[len(old)-1] + *h = old[:len(old)-1] + return last +} diff --git a/pkg/turbine/repairsim/sim_test.go b/pkg/turbine/repairsim/sim_test.go new file mode 100644 index 000000000..fc161768f --- /dev/null +++ b/pkg/turbine/repairsim/sim_test.go @@ -0,0 +1,254 @@ +package repairsim + +import ( + "reflect" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/turbine" +) + +func testLedger(t *testing.T, slots, fecs int) *Ledger { + t.Helper() + ledger, err := GenerateLedger(LedgerConfig{ + StartSlot: 20_000, + Slots: slots, + FECsPerSlot: fecs, + Seed: 7, + ShredVersion: 11, + ReferenceTick: 63, + }) + if err != nil { + t.Fatal(err) + } + return ledger +} + +func deterministicConfig(scenario Scenario) Config { + cfg := DefaultConfig(scenario) + cfg.RepairLatency = 10 * time.Millisecond + cfg.RepairJitter = time.Millisecond + cfg.DuplicateProbability = 0 + cfg.BandwidthBytesPerSec = 0 + cfg.MaxConcurrent = 32 + cfg.Seed = 19 + return cfg +} + +func TestGenerateLedgerHasExactFECCountAndValidPackets(t *testing.T) { + ledger := testLedger(t, 2, 3) + if ledger.Config.EntriesPerSlot == 0 { + t.Fatal("entry count was not resolved") + } + for _, slot := range ledger.Slots { + if len(slot.FECs) != 3 { + t.Fatalf("slot %d FECs=%d, want 3", slot.Number, len(slot.FECs)) + } + for _, fec := range slot.FECs { + if len(fec.Data) != 32 || len(fec.Coding) != 32 { + t.Fatalf("slot %d FEC %d shape=%d+%d", slot.Number, fec.Index, len(fec.Data), len(fec.Coding)) + } + for _, packet := range append(append([]Packet(nil), fec.Data...), fec.Coding...) { + shred, err := parseAndVerify(packet, ledger) + if err != nil { + t.Fatalf("slot %d FEC %d: %v", slot.Number, fec.Index, err) + } + if shred.Slot != slot.Number || shred.FECSetIndex != fec.Index { + t.Fatalf("packet routing got slot=%d FEC=%d", shred.Slot, shred.FECSetIndex) + } + } + } + } +} + +func TestNearTipRepairCompletesAndTraceIsDeterministic(t *testing.T) { + ledger := testLedger(t, 3, 2) + cfg := deterministicConfig(ScenarioNearTip) + cfg.NaturalLateShreds = true + + first, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + second, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if first.CompletedSlots != 3 || first.SpoolCompleteSlots != 3 { + t.Fatalf("completed=%d spool=%d, want 3", first.CompletedSlots, first.SpoolCompleteSlots) + } + if first.LocallyRecoveredDataShreds == 0 || first.FECDecodes == 0 { + t.Fatalf("recovered=%d decodes=%d, want both nonzero", first.LocallyRecoveredDataShreds, first.FECDecodes) + } + if first.CanceledOrLateResponses == 0 { + t.Fatal("natural late-shred scenario did not produce canceled/late repair work") + } + if !reflect.DeepEqual(first.Trace, second.Trace) { + t.Fatal("same seed/config produced different logical traces") + } +} + +func TestNearTipWithoutRepairRemainsIncomplete(t *testing.T) { + ledger := testLedger(t, 2, 2) + cfg := deterministicConfig(ScenarioNearTip) + cfg.RepairEnabled = false + cfg.NaturalLateShreds = false + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != 0 || result.SpoolCompleteSlots != 0 { + t.Fatalf("completed=%d spool=%d without repair", result.CompletedSlots, result.SpoolCompleteSlots) + } +} + +func TestNaturalLateShredsWithoutRepair(t *testing.T) { + for _, tc := range []struct { + name string + fecs int + wantCompleted int + }{ + {name: "live-arrivals-complete-slots", fecs: 1, wantCompleted: 2}, + {name: "live-arrivals-leave-other-holes", fecs: 2, wantCompleted: 0}, + } { + t.Run(tc.name, func(t *testing.T) { + ledger := testLedger(t, 2, tc.fecs) + cfg := deterministicConfig(ScenarioNearTip) + cfg.RepairEnabled = false + cfg.NaturalLateShreds = true + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != tc.wantCompleted || result.SpoolCompleteSlots != tc.wantCompleted { + t.Fatalf("completed=%d spool=%d, want %d", result.CompletedSlots, result.SpoolCompleteSlots, tc.wantCompleted) + } + // Each live arrival provides the 32nd shard in its slot's first FEC, + // recovering one missing data shred without any repair response. + if result.LocallyRecoveredDataShreds != 2 { + t.Fatalf("recovered=%d, want 2 after both live arrivals", result.LocallyRecoveredDataShreds) + } + if result.RepairRequests != 0 || result.RepairResponses != 0 || result.RepairBytesRequested != 0 { + t.Fatalf("repair disabled: requests=%d responses=%d bytes requested=%d", result.RepairRequests, result.RepairResponses, result.RepairBytesRequested) + } + if result.LogicalElapsed < cfg.RepairLatency/2+time.Microsecond { + t.Fatalf("elapsed=%v, stopped before the last live arrival", result.LogicalElapsed) + } + }) + } +} + +func TestCompleteDeliveryEstablishesZeroRepairBaseline(t *testing.T) { + ledger := testLedger(t, 2, 2) + cfg := deterministicConfig(ScenarioNearTip) + cfg.Availability = AvailabilityComplete + cfg.RepairEnabled = false + cfg.NaturalLateShreds = false + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != 2 { + t.Fatalf("completed=%d, want 2", result.CompletedSlots) + } + if result.RepairRequests != 0 || result.LocallyRecoveredDataShreds != 0 { + t.Fatalf("baseline requests=%d recovered=%d, want zero", result.RepairRequests, result.LocallyRecoveredDataShreds) + } + if result.ShredEd25519Verifications != 4 || result.ShredSignatureCacheHits != 124 { + t.Fatalf("signature cache verifies=%d hits=%d, want 4/124 for four FEC roots", + result.ShredEd25519Verifications, result.ShredSignatureCacheHits) + } +} + +func TestDeepCatchupMixedUsesThresholdRecovery(t *testing.T) { + ledger := testLedger(t, 8, 2) + cfg := deterministicConfig(ScenarioDeepCatchup) + cfg.Availability = AvailabilityMixed + cfg.NaturalLateShreds = false + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != len(ledger.Slots) { + t.Fatalf("completed=%d, want %d", result.CompletedSlots, len(ledger.Slots)) + } + if result.LocallyRecoveredDataShreds <= result.UsefulNetworkDataShreds { + t.Fatalf("local recovery=%d, network data=%d; mixed threshold scenario should recover most losses locally", result.LocallyRecoveredDataShreds, result.UsefulNetworkDataShreds) + } + if result.RepairRequests == 0 || result.RepairBytesReceived == 0 { + t.Fatal("deep catch-up completed without exercising repair") + } +} + +func TestCorruptRepairResponseRejectedThenRetried(t *testing.T) { + ledger := testLedger(t, 1, 1) + cfg := deterministicConfig(ScenarioDeepCatchup) + cfg.Availability = AvailabilityMixed + cfg.CorruptResponses = 1 + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != 1 { + t.Fatalf("completed=%d, want 1", result.CompletedSlots) + } + if result.RejectedCorruptResponses != 1 { + t.Fatalf("rejected corrupt=%d, want 1", result.RejectedCorruptResponses) + } + if result.RepairRequests < 2 { + t.Fatalf("requests=%d, want retry after corruption", result.RepairRequests) + } +} + +func parseAndVerify(packet Packet, ledger *Ledger) (*turbine.Shred, error) { + shred, err := turbine.ParseShred(packet.Bytes) + if err != nil { + return nil, err + } + if err := shred.VerifySignature(ledger.LeaderPub); err != nil { + return nil, err + } + return shred, nil +} + +func BenchmarkScenarios(b *testing.B) { + ledger, err := GenerateLedger(LedgerConfig{ + StartSlot: 30_000, Slots: 8, FECsPerSlot: 2, Seed: 23, ShredVersion: 1, ReferenceTick: 63, + }) + if err != nil { + b.Fatal(err) + } + tests := []struct { + name string + scenario Scenario + availability Availability + }{ + {name: "near-tip", scenario: ScenarioNearTip, availability: AvailabilityNearLoss}, + {name: "deep-mixed", scenario: ScenarioDeepCatchup, availability: AvailabilityMixed}, + {name: "deep-sparse", scenario: ScenarioDeepCatchup, availability: AvailabilitySparse}, + } + for _, tt := range tests { + b.Run(tt.name, func(b *testing.B) { + cfg := DefaultConfig(tt.scenario) + cfg.Availability = tt.availability + cfg.RepairLatency = 0 + cfg.RepairJitter = 0 + cfg.DuplicateProbability = 0 + cfg.BandwidthBytesPerSec = 0 + cfg.NaturalLateShreds = false + cfg.CollectTrace = false + b.ReportAllocs() + for i := 0; i < b.N; i++ { + result, err := Run(ledger, cfg) + if err != nil { + b.Fatal(err) + } + if result.CompletedSlots != len(ledger.Slots) { + b.Fatalf("completed=%d", result.CompletedSlots) + } + b.ReportMetric(float64(result.RepairRequests), "repair-requests/op") + b.ReportMetric(float64(result.LocallyRecoveredDataShreds), "recovered-shreds/op") + } + }) + } +} diff --git a/pkg/turbine/retention_sweep_test.go b/pkg/turbine/retention_sweep_test.go new file mode 100644 index 000000000..840640461 --- /dev/null +++ b/pkg/turbine/retention_sweep_test.go @@ -0,0 +1,137 @@ +package turbine + +import ( + "errors" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func sweepForTest(a *SlotAssembler) { + a.mu.Lock() + a.pruneOldSlotsLocked() + a.mu.Unlock() +} + +func TestRetentionSweepFloorMovesWithoutNewShreds(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + a.SetRetentionFloor(1000) + a.slots[1000] = &slotState{slot: 1000} + a.slots[1001] = &slotState{slot: 1001} + a.completedSlots[1001] = struct{}{} + a.SetKnownAlpenglowBlockID(1001, solana.Hash{1}) + a.PrioritizeRepairSlot(1000) + sweepForTest(a) + require.Contains(t, a.slots, uint64(1000)) + a.SetRetentionFloor(1001) + sweepForTest(a) + require.NotContains(t, a.slots, uint64(1000)) + require.NotContains(t, a.priorityRepairSlots, uint64(1000)) + require.Contains(t, a.slots, uint64(1001)) + a.SetRetentionFloor(0) + sweepForTest(a) + require.Empty(t, a.slots) + require.Empty(t, a.completedSlots) + require.Empty(t, a.knownBlockIDs) + + // Lowering the floor must permit new old repair state again. + a.SetRetentionFloor(1000) + a.slotState(1000, 1) + sweepForTest(a) + require.Contains(t, a.slots, uint64(1000)) +} + +func TestRetentionSweepReleasesCompletingProtectionAtFixedEdge(t *testing.T) { + for _, outcome := range []string{"abort", "cancel", "error", "complete", "reset"} { + t.Run(outcome, func(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + s := &slotState{slot: 1000, parentSlot: 999, completing: true, shreds: map[uint32]*Shred{0: {}}} + a.slots[s.slot] = s + a.SetKnownAlpenglowBlockID(999, solana.Hash{1}) + a.SetKnownAlpenglowBlockID(1000, solana.Hash{2}) + a.RejectAlpenglowBlockID(1000, solana.Hash{3}) + sweepForTest(a) + require.Contains(t, a.slots, uint64(1000)) + require.Contains(t, a.knownBlockIDs, uint64(999)) + require.Contains(t, a.rejectedBlockIDs, uint64(1000)) + work := &slotCompletionWork{state: s} + switch outcome { + case "abort": + a.abortCompletion(work) + case "cancel": + _, err := a.finalizeCompletion(work, processedSlotCompletion{canceled: true}) + require.NoError(t, err) + case "error": + _, err := a.finalizeCompletion(work, processedSlotCompletion{err: errors.New("decode failure")}) + require.Error(t, err) + case "complete": + _, err := a.finalizeCompletion(work, processedSlotCompletion{block: &block.Block{Slot: 1000}}) + require.NoError(t, err) + case "reset": + a.ResetSlot(1000) + } + sweepForTest(a) + require.Empty(t, a.slots) + require.Empty(t, a.completedSlots) + require.Empty(t, a.knownBlockIDs) + require.Empty(t, a.rejectedBlockIDs) + require.Empty(t, a.partialShredObs) + }) + } +} + +func TestRetentionSweepOldHintsAddedAtFixedEdge(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + sweepForTest(a) + a.SetKnownAlpenglowBlockID(1000, solana.Hash{1}) + a.RejectAlpenglowBlockID(1001, solana.Hash{2}) + sweepForTest(a) + require.Empty(t, a.knownBlockIDs) + require.Empty(t, a.rejectedBlockIDs) + a.mu.Lock() + a.trackBlockIDLocked(&block.Block{Slot: 1002, HasAlpenglowBlockID: true, AlpenglowBlockID: solana.Hash{3}}) + a.mu.Unlock() + sweepForTest(a) + require.Empty(t, a.knownBlockIDs) +} + +func TestRetentionSweepCapacityAtFixedEdge(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + a.SetRetentionFloor(1000) + a.PrioritizeRepairSlot(1001) + sweepForTest(a) + for i := 0; i < maxRetainedIncompleteSlotCap+2; i++ { + a.slotState(1000+uint64(i), 1) + } + sweepForTest(a) + require.Len(t, a.slots, maxRetainedIncompleteSlotCap) + require.Contains(t, a.slots, uint64(1000)) + require.Contains(t, a.slots, uint64(1001)) + require.NotContains(t, a.slots, uint64(1000+maxRetainedIncompleteSlotCap+1)) +} + +func BenchmarkRetentionRepeatedCompletedShred(b *testing.B) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + for slot := uint64(9488); slot <= 10000; slot++ { + a.completedSlots[slot] = struct{}{} + a.knownBlockIDs[slot] = solana.Hash{1} + a.rejectedBlockIDs[slot] = map[solana.Hash]struct{}{{2}: {}} + a.partialShredObs[slot] = PartialShredObservation{DataShreds: 1} + } + sh := &Shred{Slot: 10000, Type: ShredTypeData} + // The public ingestion path still acquires the lock and performs its + // ordinary completed-slot rejection on every packet. + _, _ = a.AddShred(sh) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + _, _ = a.AddShred(sh) + } +} diff --git a/pkg/turbine/retransmit.go b/pkg/turbine/retransmit.go index d8584e6cf..dfb0a04a2 100644 --- a/pkg/turbine/retransmit.go +++ b/pkg/turbine/retransmit.go @@ -60,10 +60,17 @@ type RetransmitConfig struct { type retransmitWork struct { packet []byte - shred ShredID - leader solana.PublicKey + // storage is exclusively owned by this work until send (including retries) + // completes. Nil denotes the unpooled oversized compatibility path. + storage *retransmitPacket + shred ShredID + leader solana.PublicKey } +// Canonical shreds fit in one Solana packet. Keep the pool fixed-size rather +// than retaining arbitrary caller-provided capacities. +type retransmitPacket [packetDataSize]byte + type cachedRetransmitNodes struct { asof time.Time nodes *ClusterNodes @@ -91,6 +98,8 @@ type retransmitParentSigCache struct { } type packetBatchSender interface { + // Send borrows packet and peers only until it returns, including on errors + // or partial sends. Implementations must copy anything they retain. Send(packet []byte, peers []*net.UDPAddr) (int, error) Close() error } @@ -120,8 +129,13 @@ type Retransmitter struct { // immediately visible alongside short-send counters. sendBufferBytes int - queue chan retransmitWork - senders []packetBatchSender + queue chan retransmitWork + senders []packetBatchSender + packetPool sync.Pool + // Only guards queue admission versus shutdown; never held during crypto, + // peer selection or socket writes. A stopped queue cannot retain late work. + submitMu sync.RWMutex + stopped bool cacheMu sync.Mutex cache map[uint64]cachedRetransmitNodes @@ -283,20 +297,53 @@ func (r *Retransmitter) Run(ctx context.Context) { sender := sender go func() { defer workers.Done() + var peers [dataPlaneFanout]*net.UDPAddr for { select { case <-ctx.Done(): return case work := <-r.queue: - r.send(work, sender) + r.send(work, sender, peers[:0]) } } }() } <-ctx.Done() + r.submitMu.Lock() + r.stopped = true + r.submitMu.Unlock() workers.Wait() - for _, sender := range r.senders { - _ = sender.Close() + // Admission is closed and senders have returned their leases. Drain the + // remaining queue without closing a channel still visible to submitters. + for { + select { + case work := <-r.queue: + r.releasePacket(work.storage) + default: + for _, sender := range r.senders { + _ = sender.Close() + } + return + } + } +} + +func (r *Retransmitter) copyPacket(packet []byte) ([]byte, *retransmitPacket) { + if len(packet) > len(retransmitPacket{}) { + return append([]byte(nil), packet...), nil + } + storage, _ := r.packetPool.Get().(*retransmitPacket) + if storage == nil { + storage = new(retransmitPacket) + } + out := storage[:len(packet)] + copy(out, packet) + return out, storage +} + +func (r *Retransmitter) releasePacket(storage *retransmitPacket) { + if storage != nil { + r.packetPool.Put(storage) } } @@ -340,25 +387,37 @@ func (r *Retransmitter) SubmitFrom(packet []byte, shred *Shred, leader solana.Pu if packetSize == 0 || packetSize > len(packet) { return fmt.Errorf("turbine retransmit: invalid canonical packet size %d/%d", packetSize, len(packet)) } - out := append([]byte(nil), packet[:packetSize]...) + // Validate the signing range before borrowing storage, so all error exits + // either precede ownership or release it through send/the queue-drop path. + offset := 0 if root != nil { - offset, err := shred.retransmitterSignatureOffset() + offset, err = shred.retransmitterSignatureOffset() if err != nil { return fmt.Errorf("turbine retransmit: locate retransmitter signature: %w", err) } - if offset+ed25519.SignatureSize > len(out) { - return fmt.Errorf("turbine retransmit: retransmitter signature slice %d:%d exceeds packet size %d", offset, offset+ed25519.SignatureSize, len(out)) + if offset+ed25519.SignatureSize > packetSize { + return fmt.Errorf("turbine retransmit: retransmitter signature slice %d:%d exceeds packet size %d", offset, offset+ed25519.SignatureSize, packetSize) } + } + out, storage := r.copyPacket(packet[:packetSize]) + if root != nil { signature := ed25519.Sign(r.cfg.Identity, root[:]) copy(out[offset:offset+ed25519.SignatureSize], signature) r.resignedShreds.Add(1) } r.submitted.Add(1) + r.submitMu.RLock() + defer r.submitMu.RUnlock() + if r.stopped { + r.releasePacket(storage) + return nil + } select { - case r.queue <- retransmitWork{packet: out, shred: id, leader: leader}: + case r.queue <- retransmitWork{packet: out, storage: storage, shred: id, leader: leader}: default: r.queueDrops.Add(1) + r.releasePacket(storage) } return nil } @@ -417,9 +476,13 @@ func (r *Retransmitter) verifyParentSignature(shred *Shred, leader solana.Public return &root, nil } -func (r *Retransmitter) send(work retransmitWork, sender packetBatchSender) { +func (r *Retransmitter) send(work retransmitWork, sender packetBatchSender, scratch []*net.UDPAddr) { + defer r.releasePacket(work.storage) + // Clear even the unused tail after selection: worker scratch must not pin + // old snapshot addresses after a topology refresh or a no-peer/error path. + defer clear(scratch[:cap(scratch)]) nodes := r.clusterNodesForSlot(work.shred.Slot) - distance, peers, err := nodes.RetransmitPeers(work.leader, work.shred, dataPlaneFanout) + distance, peers, err := nodes.retransmitPeersInto(work.leader, work.shred, dataPlaneFanout, scratch) if errors.Is(err, ErrRetransmitLoopback) { r.loopbacks.Add(1) return diff --git a/pkg/turbine/retransmit_allocation_bench_test.go b/pkg/turbine/retransmit_allocation_bench_test.go new file mode 100644 index 000000000..3f376652a --- /dev/null +++ b/pkg/turbine/retransmit_allocation_bench_test.go @@ -0,0 +1,89 @@ +package turbine + +import ( + "context" + "crypto/ed25519" + "encoding/binary" + "fmt" + "net" + "runtime" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/gossip" + "github.com/gagliardetto/solana-go" +) + +type discardRelaySender struct{} + +func (*discardRelaySender) Send(_ []byte, peers []*net.UDPAddr) (int, error) { return len(peers), nil } +func (*discardRelaySender) Close() error { return nil } + +// Includes Submit's dedupe/copy, channel handoff, routing and sender dispatch. +// The sender performs no syscalls; this isolates relay CPU/allocations rather +// than claiming a network-throughput improvement. Authentication before Submit +// and resigned-shred crypto are covered by functional tests, not this fixture. +func BenchmarkRetransmitPipeline(b *testing.B) { + for _, count := range []int{90, 512} { + b.Run(fmt.Sprint(count), func(b *testing.B) { + nodes, leader, key := relayAllocationNodes(count, true) + r, err := newRetransmitterWithSenders(RetransmitConfig{Identity: key, Peers: &mutableTVUPeers{}, Stakes: func(uint64) map[solana.PublicKey]uint64 { return nil }, QueueDepth: 256}, []packetBatchSender{&discardRelaySender{}}) + if err != nil { + b.Fatal(err) + } + r.cache[10] = cachedRetransmitNodes{asof: time.Now().Add(time.Hour), nodes: nodes} + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { r.Run(ctx); close(done) }() + packet := make([]byte, dataPayloadSize) + shred := &Shred{Slot: 10, Type: ShredTypeData, Payload: packet} + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + // Exactly one producer; leave capacity before submitting so every + // iteration measures actual forwarding rather than dropped work. + for len(r.queue) == cap(r.queue) { + runtime.Gosched() + } + binary.LittleEndian.PutUint64(packet[:8], uint64(i)) + shred.Index = uint32(i) + if err := r.Submit(packet, shred, leader, false); err != nil { + b.Fatal(err) + } + } + for { + var n uint64 + for i := range r.rootDistance { + n += r.rootDistance[i].Load() + } + if n >= uint64(b.N) { + break + } + runtime.Gosched() + } + cancel() + <-done + b.StopTimer() + if r.queueDrops.Load() != 0 { + b.Fatal("unexpected queue drop") + } + }) + } +} + +func relayAllocationNodes(count int, chacha8 bool) (*ClusterNodes, solana.PublicKey, ed25519.PrivateKey) { + key := ed25519.NewKeyFromSeed(make([]byte, ed25519.SeedSize)) + var self solana.PublicKey + copy(self[:], key.Public().(ed25519.PublicKey)) + leader := solana.PublicKey{255} + peers := make([]gossip.TVUPeer, 0, count) + stakes := map[solana.PublicKey]uint64{self: 100, leader: 50} + for i := 0; i < count; i++ { + var pub solana.PublicKey + binary.LittleEndian.PutUint64(pub[:], uint64(i+1)) + addr := &net.UDPAddr{IP: net.IPv4(127, 1, byte(i/250), byte(i%250+1)), Port: 8001} + peers = append(peers, gossip.TVUPeer{Pubkey: gossip.Pubkey(pub), TVUAddr: addr}) + stakes[pub] = uint64(i % 101) + } + return NewRetransmitClusterNodes(ClusterNodesConfig{Self: self, TVUPeers: peers, Stakes: stakes, UseChaCha8: chacha8}), leader, key +} diff --git a/pkg/turbine/retransmit_allocation_test.go b/pkg/turbine/retransmit_allocation_test.go new file mode 100644 index 000000000..3cced3129 --- /dev/null +++ b/pkg/turbine/retransmit_allocation_test.go @@ -0,0 +1,267 @@ +package turbine + +import ( + "bytes" + "context" + "net" + "sync" + "syscall" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Derive expected children from the full weighted permutation, independently +// of the streaming child-selection loop and its caller-owned output storage. +func expectedRelayPeers(nodes *ClusterNodes, leader solana.PublicKey, id ShredID, fanout int) (uint8, []*net.UDPAddr) { + order := nodes.retransmitShuffle(leader, id) + self := -1 + for i, index := range order { + if nodes.nodes[index].pubkey == nodes.selfPubkey { + self = i + break + } + } + if self < 0 { + return maxTurbineHops - 1, nil + } + offset := 0 + if self > 0 { + offset = (self - 1) % fanout + } + step := fanout + if self == 0 { + step = 1 + } + var peers []*net.UDPAddr + for position, n := (self-offset)*fanout+offset+1, 0; position < len(order) && n < fanout; position, n = position+step, n+1 { + node := nodes.nodes[order[position]] + if node.hasContact { + if addr, ok := broadcastTVUUDP(node.tvuAddr); ok { + peers = append(peers, addr) + } + } + } + return turbineRootDistance(self, fanout), peers +} + +func TestRetransmitScratchMatchesPermutation(t *testing.T) { + for _, chacha8 := range []bool{false, true} { + for _, count := range []int{0, 31, 90, 512} { + nodes, leader, _ := relayAllocationNodes(count, chacha8) + // Exercise absent contacts and unroutable peers without changing stake order. + for i := range nodes.nodes { + if i%13 == 0 { + nodes.nodes[i].hasContact = false + } + if i%17 == 0 { + nodes.nodes[i].tvuAddr = &net.UDPAddr{IP: net.ParseIP("::1"), Port: 8001} + } + } + for _, fanout := range []int{1, 3, 200} { + var scratch [dataPlaneFanout]*net.UDPAddr + for index := uint32(0); index < 40; index++ { + id := ShredID{Slot: uint64(10 + index/4), Index: index, Type: ShredType(index % 2)} + wantDistance, want := expectedRelayPeers(nodes, leader, id, fanout) + distance, got, err := nodes.retransmitPeersInto(leader, id, fanout, scratch[:0]) + require.NoError(t, err) + require.Equal(t, wantDistance, distance) + require.Equal(t, len(want), len(got)) + for i := range want { + require.Same(t, want[i], got[i]) + } + _, owned, err := nodes.RetransmitPeers(leader, id, fanout) + require.NoError(t, err) + copyOfOwned := append([]*net.UDPAddr(nil), owned...) + clear(scratch[:]) + require.Equal(t, copyOfOwned, append([]*net.UDPAddr(nil), owned...)) + } + } + } + } +} + +func TestRetransmitScratchConcurrent(t *testing.T) { + nodes, leader, _ := relayAllocationNodes(300, true) + var wg sync.WaitGroup + for range 8 { + wg.Go(func() { + var scratch [dataPlaneFanout]*net.UDPAddr + for index := uint32(0); index < 50; index++ { + id := ShredID{Slot: 10, Index: index, Type: ShredTypeData} + wd, wp := expectedRelayPeers(nodes, leader, id, 3) + d, p, err := nodes.retransmitPeersInto(leader, id, 3, scratch[:0]) + if err != nil || d != wd || len(p) != len(wp) { + t.Error("concurrent routing mismatch") + return + } + for i := range p { + if p[i] != wp[i] { + t.Error("concurrent peer mismatch") + return + } + } + } + }) + } + wg.Wait() +} + +func TestRetransmitPacketPoolOwnership(t *testing.T) { + r := &Retransmitter{} + input := bytes.Repeat([]byte{7}, packetDataSize) + a, ownerA := r.copyPacket(input) + b, ownerB := r.copyPacket(input) + require.NotSame(t, ownerA, ownerB) + clear(input) + require.Equal(t, byte(7), a[0]) + require.Equal(t, byte(7), b[0]) + r.releasePacket(ownerA) + for range 50 { + p, owner := r.copyPacket(bytes.Repeat([]byte{9}, 1203)) + clear(p) + r.releasePacket(owner) + } + require.Equal(t, bytes.Repeat([]byte{7}, packetDataSize), b) + r.releasePacket(ownerB) + oversized := bytes.Repeat([]byte{4}, packetDataSize+1) + p, owner := r.copyPacket(oversized) + require.Nil(t, owner) + clear(oversized) + require.Equal(t, byte(4), p[0]) +} + +func TestRetransmitQueuedPacketOwnsInput(t *testing.T) { + nodes, leader, key := relayAllocationNodes(90, true) + r, err := newRetransmitterWithSenders(RetransmitConfig{Identity: key, Peers: &mutableTVUPeers{}, Stakes: func(uint64) map[solana.PublicKey]uint64 { return nil }, QueueDepth: 1}, []packetBatchSender{newCaptureBatchSender()}) + require.NoError(t, err) + r.cache[10] = cachedRetransmitNodes{asof: time.Now(), nodes: nodes} + packet := bytes.Repeat([]byte{3}, dataPayloadSize) + packet[0] = 1 + shred := &Shred{Slot: 10, Index: 1, Type: ShredTypeData, Payload: packet} + require.NoError(t, r.Submit(packet, shred, leader, false)) + want := append([]byte(nil), packet...) + for i := 2; i < 20; i++ { + packet[0] = byte(i) + shred.Index = uint32(i) + require.NoError(t, r.Submit(packet, shred, leader, false)) + } + require.Equal(t, uint64(18), r.queueDrops.Load()) + clear(packet) + work := <-r.queue + require.Equal(t, want, work.packet) + // Cancellation leaves no owned copies queued after workers finish. + r.queue <- work + ctx, cancel := context.WithCancel(context.Background()) + cancel() + r.Run(ctx) + require.Empty(t, r.queue) + packet[0] = 99 + shred.Index = 99 + require.NoError(t, r.Submit(packet, shred, leader, false)) + require.Empty(t, r.queue, "late submit retained storage after workers stopped") +} + +type borrowedRelaySender struct { + t *testing.T + want []byte + entered chan struct{} + resume chan struct{} + calls int +} + +func (s *borrowedRelaySender) Send(packet []byte, peers []*net.UDPAddr) (int, error) { + s.calls++ + if s.calls == 1 { + close(s.entered) + <-s.resume + } + require.Equal(s.t, s.want, packet) + if s.calls == 1 { + return 0, syscall.EAGAIN + } + return len(peers), nil +} +func (*borrowedRelaySender) Close() error { return nil } + +func TestRetransmitPoolLeaseSurvivesSendRetries(t *testing.T) { + nodes, leader, key := relayAllocationNodes(31, true) + id := ShredID{Slot: 10, Type: ShredTypeData} + // Select a shred for which this validator is the root and has children. + for ; id.Index < 10000; id.Index++ { + d, p, err := nodes.RetransmitPeers(leader, id, 200) + require.NoError(t, err) + if d == 0 && len(p) > 1 { + break + } + } + require.Less(t, id.Index, uint32(10000)) + want := bytes.Repeat([]byte{7}, dataPayloadSize) + sender := &borrowedRelaySender{t: t, want: want, entered: make(chan struct{}), resume: make(chan struct{})} + r, err := newRetransmitterWithSenders(RetransmitConfig{Identity: key, Peers: &mutableTVUPeers{}, Stakes: func(uint64) map[solana.PublicKey]uint64 { return nil }}, []packetBatchSender{sender}) + require.NoError(t, err) + r.cache[10] = cachedRetransmitNodes{asof: time.Now(), nodes: nodes} + packet, owner := r.copyPacket(want) + var scratch [dataPlaneFanout]*net.UDPAddr + done := make(chan struct{}) + go func() { + defer close(done) + r.send(retransmitWork{packet: packet, storage: owner, shred: id, leader: leader}, sender, scratch[:0]) + }() + select { + case <-sender.entered: + case <-time.After(5 * time.Second): + t.Fatal("send did not start") + } + for range 100 { + p, owned := r.copyPacket(want) + clear(p) + r.releasePacket(owned) + } + close(sender.resume) + select { + case <-done: + case <-time.After(5 * time.Second): + t.Fatal("send did not finish") + } + require.Equal(t, 2, sender.calls) + for _, addr := range scratch { + require.Nil(t, addr, "scratch pinned prior snapshot") + } +} + +func TestRetransmitConcurrentSubmitAndStop(t *testing.T) { + nodes, leader, key := relayAllocationNodes(90, true) + r, err := newRetransmitterWithSenders(RetransmitConfig{Identity: key, Peers: &mutableTVUPeers{}, Stakes: func(uint64) map[solana.PublicKey]uint64 { return nil }, QueueDepth: 8}, []packetBatchSender{&discardRelaySender{}, &discardRelaySender{}}) + require.NoError(t, err) + r.cache[10] = cachedRetransmitNodes{asof: time.Now(), nodes: nodes} + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { r.Run(ctx); close(done) }() + var wg sync.WaitGroup + for worker := 0; worker < 8; worker++ { + wg.Go(func() { + packet := make([]byte, dataPayloadSize) + for i := 0; i < 100; i++ { + packet[0], packet[1] = byte(worker), byte(i) + shred := &Shred{Slot: 10, Index: uint32(worker*100 + i), Type: ShredTypeData, Payload: packet} + if err := r.Submit(packet, shred, leader, false); err != nil { + t.Error(err) + } + if i == 50 { + cancel() + } + } + }) + } + wg.Wait() + cancel() + select { + case <-done: + case <-time.After(5 * time.Second): + t.Fatal("shutdown blocked") + } + require.Empty(t, r.queue) +} diff --git a/pkg/turbine/shred.go b/pkg/turbine/shred.go index 4361f7cbd..23a4d002a 100644 --- a/pkg/turbine/shred.go +++ b/pkg/turbine/shred.go @@ -457,13 +457,12 @@ func merkleHashNode(left []byte, right []byte) solana.Hash { return hashv([][]byte{[]byte(merkleHashPrefixNode), left, right}) } -func hashv(parts [][]byte) solana.Hash { +func hashv(parts [][]byte) (out solana.Hash) { h := sha256.New() for _, part := range parts { _, _ = h.Write(part) } - var out solana.Hash - copy(out[:], h.Sum(nil)) + _ = h.Sum(out[:0]) return out } diff --git a/pkg/turbine/shredspool.go b/pkg/turbine/shredspool.go index 8f8003eb1..cebb0c837 100644 --- a/pkg/turbine/shredspool.go +++ b/pkg/turbine/shredspool.go @@ -12,6 +12,8 @@ import ( "strconv" "strings" "sync" + + "github.com/Overclock-Validator/mithril/pkg/mlog" ) // ShredSpool is a disposable on-disk cache of VERIFIED raw shreds, one @@ -32,25 +34,28 @@ import ( // re-fetch near the tip, the low end borders replay and is what repair would // otherwise pay for dearly. type ShredSpool struct { - mu sync.Mutex - dir string - open map[uint64]*spoolFile - sizes map[uint64]int64 // per-slot bytes on disk (open writers included) - seen map[uint64]map[spoolShredKey]struct{} // distinct shreds appended this run - validated map[uint64]bool // adopted files whose record tail was checked this run - complete map[uint64]SpoolSlotMeta - journal *os.File // append-only completeness journal (complete.idx) - bytes int64 - maxBytes int64 - highestSlot uint64 - haveHighest bool - floor uint64 - closed bool + mu sync.Mutex + dir string + open map[uint64]*spoolFile + sizes map[uint64]int64 // per-slot bytes on disk (open writers included) + seen map[uint64]map[spoolShredKey]struct{} // distinct shreds appended this run + validated map[uint64]bool // adopted files whose record tail was checked this run + complete map[uint64]SpoolSlotMeta + journal *spoolCompletionJournal // ordered completeness hints (complete.idx) + journalOverflow bool // Close must retry the current hints after queue overflow + bytes int64 + maxBytes int64 + highestSlot uint64 + haveHighest bool + floor uint64 + closed bool } // SpoolSlotMeta records a slot proven FULLY assembled: every data shred // 0..LastIndex was held when the assembler completed it. The completeness -// index is what turns the spool from a byte cache into the seed of a +// index is a repair hint, not proof that all buffered packets survived a crash. +// The assembler still validates coverage when hydrating a slot. This turns +// the spool from a byte cache into the seed of a // repair-serving shredstore: complete slots need zero network on restart, // answer HighestWindowIndex honestly, and define the serving/retention set. type SpoolSlotMeta struct { @@ -170,10 +175,10 @@ func (s *ShredSpool) loadJournal() { if err != nil { return // journal unavailable: completeness degrades to per-run only } + s.journal = newSpoolCompletionJournal(f) for slot, meta := range s.complete { - f.Write(spoolJournalRecord(slot, meta)) + s.journal.complete(slot, meta) } - s.journal = f } func spoolJournalRecord(slot uint64, meta SpoolSlotMeta) []byte { @@ -186,11 +191,13 @@ func spoolJournalRecord(slot uint64, meta SpoolSlotMeta) []byte { // MarkComplete records that the slot fully assembled (data shreds // 0..lastIndex all held). Called by the assembler's completion hook, so -// hydrating an adopted file re-marks it for free. Idempotent. +// hydrating an adopted file re-marks it for free. Idempotent. Journal submission +// never waits for storage or queue space; a crash may lose this repair hint, +// causing reassembly/repair, but cannot authorize a vote or advance a checkpoint. func (s *ShredSpool) MarkComplete(slot uint64, lastIndex uint32, shreds uint32) { s.mu.Lock() defer s.mu.Unlock() - if slot < s.floor || shreds == 0 { + if s.closed || slot < s.floor || shreds == 0 { return } if _, done := s.complete[slot]; done { @@ -198,8 +205,8 @@ func (s *ShredSpool) MarkComplete(slot uint64, lastIndex uint32, shreds uint32) } meta := SpoolSlotMeta{LastIndex: lastIndex, Shreds: shreds} s.complete[slot] = meta - if s.journal != nil { - s.journal.Write(spoolJournalRecord(slot, meta)) + if s.journal != nil && !s.journal.tryComplete(slot, meta) { + s.journalOverflow = true } } @@ -371,12 +378,18 @@ func (s *ShredSpool) ensureRoomLocked(slot uint64, additional int64) bool { if slot >= s.highestSlot { return false } - s.dropSlotLocked(s.highestSlot) + if !s.dropSlotLocked(s.highestSlot) { + return false + } } return true } -func (s *ShredSpool) dropSlotLocked(slot uint64) { +func (s *ShredSpool) dropSlotLocked(slot uint64) bool { + if err := s.invalidateCompleteLocked(slot); err != nil { + mlog.Log.Warnf("shred spool: retaining slot %d after completion invalidation failed: %v", slot, err) + return false + } s.closeSlotLocked(slot) s.bytes -= s.sizes[slot] delete(s.sizes, slot) @@ -384,14 +397,20 @@ func (s *ShredSpool) dropSlotLocked(slot uint64) { delete(s.validated, slot) delete(s.complete, slot) _ = os.Remove(s.pathFor(slot)) - if s.journal != nil { - // Supersede any older completion record if this slot number is later - // re-created before the journal is compacted on restart. - s.journal.Write(spoolJournalRecord(slot, SpoolSlotMeta{})) - } if s.haveHighest && slot == s.highestSlot { s.recomputeHighestLocked() } + return true +} + +// Remove the live hint immediately, but do not mutate its file until older +// journal hints have been superseded. A failed fence leaves the file intact. +func (s *ShredSpool) invalidateCompleteLocked(slot uint64) error { + delete(s.complete, slot) + if s.journal != nil { + return s.journal.invalidate(slot) + } + return nil } func (s *ShredSpool) recomputeHighestLocked() { @@ -407,7 +426,9 @@ func (s *ShredSpool) recomputeHighestLocked() { // DiscardSlot removes both persisted packets and the completeness marker for // a poisoned or rejected slot. The journal tombstone prevents an older // completion record from being resurrected if repair immediately recreates a -// partial file with the same slot number. +// partial file with the same slot number. If the journal cannot durably fence +// the old marker, retain the old file and reject replacement writes until a +// retry succeeds; deleting it would permit stale completeness to certify new data. func (s *ShredSpool) DiscardSlot(slot uint64) { if s == nil { return @@ -483,16 +504,15 @@ func (s *ShredSpool) readSlotLocked(slot uint64) ([][]byte, error) { validEnd = packetEnd } if validEnd != len(data) { + if err := s.invalidateCompleteLocked(slot); err != nil { + return nil, err + } if err := os.Truncate(path, int64(validEnd)); err != nil { return nil, fmt.Errorf("truncate corrupt shred spool tail for slot %d: %w", slot, err) } oldSize := s.sizes[slot] s.sizes[slot] = int64(validEnd) s.bytes += int64(validEnd) - oldSize - delete(s.complete, slot) - if s.journal != nil { - s.journal.Write(spoolJournalRecord(slot, SpoolSlotMeta{})) - } } s.validated[slot] = true return packets, nil @@ -533,12 +553,20 @@ func (s *ShredSpool) Stats() (slots int, bytes int64) { func (s *ShredSpool) Close() { s.mu.Lock() defer s.mu.Unlock() + if s.closed { + return + } s.closed = true for slot := range s.open { s.closeSlotLocked(slot) } if s.journal != nil { - _ = s.journal.Close() + if s.journalOverflow { + for slot, meta := range s.complete { + s.journal.complete(slot, meta) + } + } + s.journal.close() s.journal = nil } } diff --git a/pkg/turbine/shredspool_benchmark_test.go b/pkg/turbine/shredspool_benchmark_test.go new file mode 100644 index 000000000..d3e30cf6f --- /dev/null +++ b/pkg/turbine/shredspool_benchmark_test.go @@ -0,0 +1,37 @@ +package turbine + +import ( + "path/filepath" + "sort" + "strconv" + "testing" + "time" +) + +// Run this identical public-API benchmark on both revisions. Each sample owns +// a fresh spool, so no completed-slot dedupe or full-queue dropping is timed. +// Close is untimed but drains the writer before the next sample. These ordinary +// filesystem measurements do not simulate rare storage stalls or whole replay. +func BenchmarkShredSpoolMarkComplete(b *testing.B) { + root := b.TempDir() + elapsed := make([]int64, 0, b.N) + b.ReportAllocs() + for i := 0; i < b.N; i++ { + b.StopTimer() + s, err := OpenShredSpool(filepath.Join(root, strconv.Itoa(i)), 0) + if err != nil { + b.Fatal(err) + } + s.Append(100, []byte("packet")) + b.StartTimer() + start := time.Now() + s.MarkComplete(100, 0, 1) + duration := time.Since(start).Nanoseconds() + b.StopTimer() + elapsed = append(elapsed, duration) + s.Close() + } + sort.Slice(elapsed, func(i, j int) bool { return elapsed[i] < elapsed[j] }) + b.ReportMetric(float64(elapsed[(len(elapsed)-1)/2]), "p50-ns") + b.ReportMetric(float64(elapsed[(99*len(elapsed)+99)/100-1]), "p99-ns") +} diff --git a/pkg/turbine/shredspool_journal.go b/pkg/turbine/shredspool_journal.go new file mode 100644 index 000000000..a537e202f --- /dev/null +++ b/pkg/turbine/shredspool_journal.go @@ -0,0 +1,102 @@ +package turbine + +import ( + "fmt" + "io" +) + +// Completion records are repair-cache hints, not voting or checkpoint state. +// A bounded writer removes their disk I/O from block delivery. Only completion +// hints may be dropped when the queue is full; live completeness stays in memory +// and Close retries the current hints before the next opener takes ownership. +const spoolJournalQueueSize = 256 + +type spoolJournalFile interface { + io.Writer + Truncate(int64) error + Close() error +} + +type spoolJournalRequest struct { + record [spoolJournalRecordSize]byte + done chan error // non-nil for an invalidation that must precede file mutation +} + +type spoolCompletionJournal struct { + requests chan spoolJournalRequest + done chan struct{} +} + +func newSpoolCompletionJournal(file spoolJournalFile) *spoolCompletionJournal { + j := &spoolCompletionJournal{requests: make(chan spoolJournalRequest, spoolJournalQueueSize), done: make(chan struct{})} + go j.run(file) + return j +} + +func journalRequest(slot uint64, meta SpoolSlotMeta) spoolJournalRequest { + var req spoolJournalRequest + copy(req.record[:], spoolJournalRecord(slot, meta)) + return req +} + +// The spool mutex serializes submissions and Close, but the worker never takes +// that mutex. A stalled write therefore cannot directly stall MarkComplete. +func (j *spoolCompletionJournal) tryComplete(slot uint64, meta SpoolSlotMeta) bool { + select { + case j.requests <- journalRequest(slot, meta): + return true + default: + return false + } +} + +func (j *spoolCompletionJournal) complete(slot uint64, meta SpoolSlotMeta) { + j.requests <- journalRequest(slot, meta) +} + +// Never drop or reorder invalidations. Wait for earlier completions and this +// tombstone before deleting/replacing/truncating a slot file. These rare paths +// may still wait for storage while holding the spool mutex. Queueing tombstones +// without this fence could resurrect an old completion after a crash. +func (j *spoolCompletionJournal) invalidate(slot uint64) error { + req := journalRequest(slot, SpoolSlotMeta{}) + req.done = make(chan error, 1) + j.requests <- req + return <-req.done +} + +func (j *spoolCompletionJournal) close() { + close(j.requests) + <-j.done +} + +func (j *spoolCompletionJournal) run(file spoolJournalFile) { + defer close(j.done) + defer file.Close() + failed, invalidated := false, false + for req := range j.requests { + var err error + if !failed { + var n int + n, err = file.Write(req.record[:]) + if err == nil && n != len(req.record) { + err = io.ErrShortWrite + } + failed = err != nil + } + if failed && !invalidated { + // Never append behind a short record, or acknowledge an invalidation + // while old completion hints remain. Empty the disposable journal and + // disable further hint writes for this opener. If even truncation fails, + // the caller must leave the slot file unchanged and retry later. + err = file.Truncate(0) + invalidated = err == nil + if err != nil { + err = fmt.Errorf("invalidate failed shred completeness journal: %w", err) + } + } + if req.done != nil { + req.done <- err + } + } +} diff --git a/pkg/turbine/shredspool_journal_test.go b/pkg/turbine/shredspool_journal_test.go new file mode 100644 index 000000000..1ba9169ce --- /dev/null +++ b/pkg/turbine/shredspool_journal_test.go @@ -0,0 +1,205 @@ +package turbine + +import ( + "bytes" + "errors" + "os" + "path/filepath" + "sync" + "testing" + "time" +) + +type gatedSpoolJournal struct { + *os.File + started chan struct{} + release chan struct{} + once sync.Once +} + +func (f *gatedSpoolJournal) Write(p []byte) (int, error) { + f.once.Do(func() { close(f.started); <-f.release }) + return f.File.Write(p) +} +func gateSpoolJournal(t *testing.T, s *ShredSpool) *gatedSpoolJournal { + t.Helper() + s.journal.close() + file, err := os.OpenFile(filepath.Join(s.dir, spoolJournalName), os.O_WRONLY|os.O_APPEND, 0) + if err != nil { + t.Fatal(err) + } + gate := &gatedSpoolJournal{File: file, started: make(chan struct{}), release: make(chan struct{})} + s.journal = newSpoolCompletionJournal(gate) + return gate +} +func waitSpoolTest(t *testing.T, done <-chan struct{}) { + t.Helper() + select { + case <-done: + case <-time.After(5 * time.Second): + t.Fatal("spool operation did not complete") + } +} + +func TestShredSpoolCompletionDoesNotWaitForJournalAndCloseDrainsOverflow(t *testing.T) { + dir := t.TempDir() + s, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + // Seed real slot files before blocking only the completeness writer. + for slot := uint64(1); slot <= spoolJournalQueueSize+8; slot++ { + s.Append(slot, []byte("packet")) + } + gate := gateSpoolJournal(t, s) + var release sync.Once + t.Cleanup(func() { release.Do(func() { close(gate.release) }); s.Close() }) + s.MarkComplete(1, 0, 1) + waitSpoolTest(t, gate.started) + done := make(chan struct{}) + go func() { + for slot := uint64(2); slot <= spoolJournalQueueSize+8; slot++ { + s.MarkComplete(slot, 0, 1) + } + close(done) + }() + waitSpoolTest(t, done) + if !s.journalOverflow { + t.Fatal("test did not overflow the bounded queue") + } + if s.CompleteSlots() != spoolJournalQueueSize+8 { + t.Fatal("overflow lost live completeness") + } + closed := make(chan struct{}) + go func() { s.Close(); close(closed) }() + select { + case <-closed: + t.Fatal("Close returned before the blocked writer drained") + default: + } + release.Do(func() { close(gate.release) }) + waitSpoolTest(t, closed) + s.Close() // idempotent; no closed-channel send + s.MarkComplete(9999, 0, 1) + if _, ok := s.IsComplete(9999); ok { + t.Fatal("post-Close completion was accepted") + } + reopened, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + defer reopened.Close() + if reopened.CompleteSlots() != spoolJournalQueueSize+8 { + t.Fatal("clean handoff lost overflowed completion hints") + } +} + +func TestShredSpoolInvalidationWaitsBeforeReplacement(t *testing.T) { + dir := t.TempDir() + s, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + s.Append(800, []byte("old-file")) + if _, err := s.ReadSlot(800); err != nil { + t.Fatal(err) + } + gate := gateSpoolJournal(t, s) + var release sync.Once + t.Cleanup(func() { release.Do(func() { close(gate.release) }); s.Close() }) + s.MarkComplete(800, 0, 1) + waitSpoolTest(t, gate.started) + done := make(chan struct{}) + go func() { s.DiscardSlot(800); s.Append(800, []byte("replacement-partial")); close(done) }() + // While the earlier completion write is stalled, replacement must wait. + select { + case <-done: + t.Fatal("replacement passed an undrained invalidation") + case <-time.After(20 * time.Millisecond): + } + data, err := os.ReadFile(s.pathFor(800)) + if err != nil || !bytes.Contains(data, []byte("old-file")) { + t.Fatalf("old file mutated before invalidation: %v", err) + } + release.Do(func() { close(gate.release) }) + waitSpoolTest(t, done) + s.Close() + reopened, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + defer reopened.Close() + if _, ok := reopened.IsComplete(800); ok { + t.Fatal("older completion resurrected for replacement") + } + packets, err := reopened.ReadSlot(800) + if err != nil || len(packets) != 1 || string(packets[0]) != "replacement-partial" { + t.Fatalf("replacement: %q %v", packets, err) + } +} + +type faultySpoolJournal struct { + *os.File + failTruncate bool // only changed while worker is fenced by invalidate's reply +} + +func (f *faultySpoolJournal) Write(p []byte) (int, error) { return f.File.Write(p[:len(p)/2]) } +func (f *faultySpoolJournal) Truncate(n int64) error { + if f.failTruncate { + return errors.New("injected truncate failure") + } + return f.File.Truncate(n) +} + +func TestShredSpoolJournalShortWriteDisablesHintsAndFencesMutations(t *testing.T) { + for _, failTruncate := range []bool{false, true} { + dir := t.TempDir() + s, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + s.Append(100, []byte("old-file")) + s.MarkComplete(100, 0, 1) + s.Close() + s, err = OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + s.journal.close() + file, err := os.OpenFile(filepath.Join(dir, spoolJournalName), os.O_WRONLY|os.O_APPEND, 0) + if err != nil { + t.Fatal(err) + } + faulty := &faultySpoolJournal{File: file, failTruncate: failTruncate} + s.journal = newSpoolCompletionJournal(faulty) + s.mu.Lock() + dropped := s.dropSlotLocked(100) + s.mu.Unlock() + if dropped == failTruncate { + t.Fatalf("drop=%v with truncate failure=%v", dropped, failTruncate) + } + if failTruncate { + data, err := os.ReadFile(s.pathFor(100)) + if err != nil || !bytes.Contains(data, []byte("old-file")) { + t.Fatal("failed journal fence mutated slot file") + } + faulty.failTruncate = false + s.DiscardSlot(100) // retry can now invalidate all old hints + } + s.Append(100, []byte("partial")) + s.MarkComplete(101, 0, 1) + s.Close() + data, err := os.ReadFile(filepath.Join(dir, spoolJournalName)) + if err != nil || len(data) != 0 { + t.Fatalf("failed journal must stay empty, got %d bytes: %v", len(data), err) + } + reopened, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + if _, ok := reopened.IsComplete(100); ok { + t.Fatal("failed journal resurrected completion") + } + reopened.Close() + } +} diff --git a/pkg/turbine/sigcache.go b/pkg/turbine/sigcache.go index 7f3b13139..990023d9b 100644 --- a/pkg/turbine/sigcache.go +++ b/pkg/turbine/sigcache.go @@ -20,7 +20,13 @@ import ( // hit reproduces exactly the result of re-running it on the same inputs. // Tampered content can never hit — different bytes yield a different root, // hence a different key. Failures are never cached. -type shredSigCache struct { +// ShredSignatureVerifier authenticates Merkle shreds with the same bounded, +// per-root result cache used by UDPReceiver. The cache never stores failures; +// each packet's Merkle proof is still evaluated before a cache lookup. +// +// It is exported so deterministic and loopback ingress harnesses can exercise +// production validation without constructing a UDPReceiver. +type ShredSignatureVerifier struct { mu sync.Mutex cur map[shredSigCacheKey]struct{} prev map[shredSigCacheKey]struct{} @@ -29,6 +35,10 @@ type shredSigCache struct { verifies atomic.Uint64 } +// Keep the internal receiver/test name as an alias; there is one +// implementation and one cache contract. +type shredSigCache = ShredSignatureVerifier + type shredSigCacheKey struct { leader solana.PublicKey root solana.Hash @@ -42,10 +52,17 @@ const shredSigCacheGenCap = 4096 // verifyShred authenticates a shred exactly like Shred.VerifySignature, with // the per-root ed25519 result cached. -func (c *shredSigCache) verifyShred(s *Shred, leader solana.PublicKey) error { +func (c *ShredSignatureVerifier) verifyShred(s *Shred, leader solana.PublicKey) error { + _, err := c.verifyShredRoot(s, leader) + return err +} + +// verifyShredRoot also returns the root authenticated for these exact bytes. +// Callers must keep the shred immutable through assembler admission. +func (c *ShredSignatureVerifier) verifyShredRoot(s *Shred, leader solana.PublicKey) (solana.Hash, error) { root, err := s.MerkleRoot() if err != nil { - return err + return solana.Hash{}, err } key := shredSigCacheKey{leader: leader, root: root, sig: s.Signature} @@ -53,28 +70,34 @@ func (c *shredSigCache) verifyShred(s *Shred, leader solana.PublicKey) error { if _, ok := c.cur[key]; ok { c.mu.Unlock() c.hits.Add(1) - return nil + return root, nil } if _, ok := c.prev[key]; ok { // Promote: a set straddling a rotation keeps its entry hot. c.addLocked(key) c.mu.Unlock() c.hits.Add(1) - return nil + return root, nil } c.mu.Unlock() c.verifies.Add(1) if !narya.VerifyStrict(leader[:], root[:], s.Signature[:]) { - return fmt.Errorf("%w: slot %d shred %d", ErrInvalidSignature, s.Slot, s.Index) + return solana.Hash{}, fmt.Errorf("%w: slot %d shred %d", ErrInvalidSignature, s.Slot, s.Index) } c.mu.Lock() c.addLocked(key) c.mu.Unlock() - return nil + return root, nil } -func (c *shredSigCache) addLocked(key shredSigCacheKey) { +// Verify authenticates one shred and retains successful root/signature tuples +// for sibling shreds in the same FEC set. +func (c *ShredSignatureVerifier) Verify(s *Shred, leader solana.PublicKey) error { + return c.verifyShred(s, leader) +} + +func (c *ShredSignatureVerifier) addLocked(key shredSigCacheKey) { if c.cur == nil { c.cur = make(map[shredSigCacheKey]struct{}, shredSigCacheGenCap) } @@ -85,6 +108,11 @@ func (c *shredSigCache) addLocked(key shredSigCacheKey) { c.cur[key] = struct{}{} } -func (c *shredSigCache) stats() (hits, verifies uint64) { +func (c *ShredSignatureVerifier) stats() (hits, verifies uint64) { return c.hits.Load(), c.verifies.Load() } + +// Stats reports cache hits and actual Ed25519 verifications. +func (c *ShredSignatureVerifier) Stats() (hits, verifies uint64) { + return c.stats() +} diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go new file mode 100644 index 000000000..9e8def5a6 --- /dev/null +++ b/pkg/turbine/stream.go @@ -0,0 +1,400 @@ +package turbine + +import ( + "context" + "errors" + "sort" + "time" + "weak" + + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// Streaming feed: the assembler already decodes and signature-verifies every +// closed DATA_COMPLETE range of a slot while its shreds arrive (the entry +// prefetch). The feed exposes those batches, in the order they become ready, +// to one subscriber — replay's streaming executor — as immutable views, so the +// block can be executed while the rest of it is still in flight. +// +// The feed is advisory. Events are wake-ups: a full subscriber channel drops +// the event, and the subscriber recovers by asking PendingStreamBatches and +// StreamStatus, which read the assembler's own state under its lock. Nothing +// here changes how a slot completes, how its identity is attached, or how the +// complete block is emitted; the complete block remains the authority. + +// StreamGeneration identifies one assembly of a slot. A slot that is reset +// and assembled again is a different generation. It is opaque: consumers +// compare it for equality and pass it back to the assembler. +type StreamGeneration struct { + slot uint64 + state *slotState +} + +// Slot returns the generation's slot. +func (g StreamGeneration) Slot() uint64 { return g.slot } + +// IsZero reports whether the generation was never set. +func (g StreamGeneration) IsZero() bool { return g.state == nil } + +// NewDetachedStreamGeneration returns a non-zero generation for slot that no +// assembler knows about. It exists so consumers (replay's streaming executor) +// can unit-test their state machines with a fake feed; a real assembler +// reports it as StreamGone. +func NewDetachedStreamGeneration(slot uint64) StreamGeneration { + return StreamGeneration{slot: slot, state: &slotState{slot: slot}} +} + +// NewDetachedStreamBatch builds a ready entry-batch view for consumers' unit +// tests: txs are its transactions and identities, when non-nil, is a +// completed verification result for exactly those transactions (as +// txverify.BatchVerifier.VerifyWithMessageIdentities produces). With nil +// identities the batch reports itself unverified. The assembler never builds +// batches this way. +func NewDetachedStreamBatch(g StreamGeneration, start, end uint32, txs []*solana.Transaction, identities []txverify.VerifiedMessageIdentity) *StreamBatch { + ready := make(chan struct{}) + close(ready) + batch := &prefetchedShredBatch{start: start, end: end, ready: ready} + if identities != nil { + batch.verification = &transactionVerification{done: ready, cancel: func() {}, identities: identities, finishedAt: time.Now()} + } + view := newStreamBatch(g, batch) + view.Transactions = txs + return view +} + +// NewDetachedStreamMarker builds a marker batch view (header, update-parent +// or footer) for consumers' unit tests. +func NewDetachedStreamMarker(g StreamGeneration, start, end uint32, kind StreamMarkerKind, parentSlot uint64, parentBlockID solana.Hash) *StreamBatch { + ready := make(chan struct{}) + close(ready) + view := newStreamBatch(g, &prefetchedShredBatch{start: start, end: end, ready: ready, marker: true}) + view.Marker = kind + view.ParentSlot = parentSlot + view.ParentBlockID = parentBlockID + return view +} + +// StreamMarkerKind classifies a batch that carries an Alpenglow block +// component instead of entries. +type StreamMarkerKind uint8 + +const ( + // StreamMarkerNone is an ordinary entry batch. + StreamMarkerNone StreamMarkerKind = iota + // StreamMarkerHeader is the block header (FEC set 0): parent slot and ID. + StreamMarkerHeader + // StreamMarkerUpdateParent selects an older parent and abandons every + // batch before ReplayFECSetIndex (the optimistic prefix). + StreamMarkerUpdateParent + // StreamMarkerFooter is the block footer (certificates, bank hash, clock). + StreamMarkerFooter +) + +// StreamBatch is an immutable view of one decoded DATA_COMPLETE range. Its +// transactions are the same objects the complete block will reference when +// completion reuses this batch (byte-identical shreds), which is what lets a +// streaming consumer prove its executed prefix is the block by identity. +type StreamBatch struct { + Slot uint64 + Generation StreamGeneration + Start, End uint32 + Marker StreamMarkerKind + // Parent fields are set for header and UpdateParent markers. + ParentSlot uint64 + ParentBlockID solana.Hash + ReplayFECSetIndex uint32 + // Transactions is empty for markers and for batches that failed to decode. + Transactions []*solana.Transaction + // Err is the decode error; a batch with Err makes the whole slot invalid. + Err error + // ReadyAt is the original publication time, preserved across polling and + // notification recovery. It is not the time a consumer looked up the batch. + ReadyAt time.Time + + batch *prefetchedShredBatch +} + +// ErrStreamBatchUnverified reports that no signature-verification result is +// attached to the batch (no transactions, or admission was refused); the +// consumer must verify signatures itself. +var ErrStreamBatchUnverified = errors.New("stream batch has no verification result") + +// WaitVerification observes the batch's asynchronous signature verification and +// returns the verifier's message identities, one per transaction, bound to +// Transactions (see block.PrepareVerifiedTransactionMessageIdentities). A +// nil error with verified == false means no result is attached and the +// caller must verify itself; any other error means a signature failed (the +// slot is invalid) or ctx ended. Unlike the owning verifier wait, a context +// timeout returns without cancelling or joining the job: turbine retains the +// immutable transaction storage and joins readers before releasing reservations. +// A caller timing out must not mutate the batch or its transactions. +func (sb *StreamBatch) WaitVerification(ctx context.Context) (identities []txverify.VerifiedMessageIdentity, verified bool, err error) { + if sb == nil || sb.batch == nil { + return nil, false, ErrStreamBatchUnverified + } + if sb.batch.verification == nil { + return nil, false, nil + } + if ctx == nil { + ctx = context.Background() + } + future := sb.batch.verification + select { + case <-ctx.Done(): + return nil, false, ctx.Err() + case <-future.done: + } + if err := ctx.Err(); err != nil { + return nil, false, err + } + if future.err != nil { + return nil, false, future.err + } + if len(sb.batch.verification.identities) != len(sb.Transactions) { + return nil, false, nil + } + return sb.batch.verification.identities, true, nil +} + +// StreamEventKind is the kind of a feed wake-up. +type StreamEventKind uint8 + +const ( + // StreamBatchReady: Batch was decoded (and its verification submitted). + StreamBatchReady StreamEventKind = iota + // StreamCancelled: the generation's state is gone without a complete + // block (reset, eviction, invalid identity, shutdown). Reason says why. + StreamCancelled + // StreamCompleted: the generation assembled and the complete block is on + // its way through the normal emission path. + StreamCompleted +) + +// StreamEvent is an advisory wake-up. Call Resolve before inspecting Generation +// or Batch. Queued notifications hold only weak references, so a stalled +// subscriber cannot retain retired slot buffers outside the prefetch budget. +type StreamEvent struct { + Kind StreamEventKind + Slot uint64 + Generation StreamGeneration + Batch *StreamBatch + Reason string + state weak.Pointer[slotState] + batch weak.Pointer[prefetchedShredBatch] +} + +// Resolve acquires ownership of a still-live notification. A false result means +// the opportunity has expired; whole-block replay remains authoritative. Resolved +// generations retain their state for polling, including after completion. Detached +// events supplied by test feeds are already resolved. +func (e StreamEvent) Resolve() (StreamEvent, bool) { + if !e.Generation.IsZero() { + return e, true + } + state := e.state.Value() + if state == nil { + return StreamEvent{}, false + } + e.Generation = StreamGeneration{slot: e.Slot, state: state} + if e.Kind == StreamBatchReady { + batch := e.batch.Value() + if batch == nil { + return StreamEvent{}, false + } + e.Batch = newStreamBatch(e.Generation, batch) + } + return e, true +} + +// StreamStatus is the assembler's view of a generation. +type StreamStatus uint8 + +const ( + // StreamActive: the generation is the slot's current assembly. + StreamActive StreamStatus = iota + // StreamDone: the generation completed and its block was (or is being) + // emitted. + StreamDone + // StreamGone: the generation was discarded without a block. + StreamGone +) + +// SubscribeStream installs the single feed subscriber. Events are sent +// without blocking; a full channel drops the event and counts it. +func (a *SlotAssembler) SubscribeStream(ch chan<- StreamEvent) { + a.mu.Lock() + defer a.mu.Unlock() + a.streamSubscriber = ch + if ch == nil { + a.streamRepairParent = nil + a.streamRepairChild = nil + a.streamRepairInvalidChild = nil + } +} + +// StreamDroppedEvents reports wake-ups dropped because the subscriber was +// full; the subscriber polls PendingStreamBatches after any wake-up, so a +// non-zero count is a sizing hint, not a correctness problem. +func (a *SlotAssembler) StreamDroppedEvents() uint64 { + a.mu.Lock() + defer a.mu.Unlock() + return a.streamDroppedEvents +} + +func (a *SlotAssembler) publishStreamLocked(event StreamEvent) { + if a.streamSubscriber == nil { + return + } + select { + case a.streamSubscriber <- event: + default: + a.streamDroppedEvents++ + } +} + +// StreamStatusOf reports whether a generation is still the slot's current +// assembly, completed into a block, or gone. +func (a *SlotAssembler) StreamStatusOf(g StreamGeneration) StreamStatus { + if g.state == nil { + return StreamGone + } + a.mu.Lock() + defer a.mu.Unlock() + return a.streamStatusLocked(g) +} + +func (a *SlotAssembler) streamStatusLocked(g StreamGeneration) StreamStatus { + if a.slots[g.slot] == g.state { + // Failed completions retain state for diagnostics. Polling must still + // see cancellation when the bounded event channel dropped its wake-up. + if g.state.streamCancelReason != "" { + return StreamGone + } + return StreamActive + } + if g.state.streamCompleted { + return StreamDone + } + return StreamGone +} + +// cancelUndeliveredStream closes a completed generation whose result was +// abandoned during receiver shutdown. Identity binding avoids cancelling a +// replacement generation assembled for the same slot. +func (a *SlotAssembler) cancelUndeliveredStream(g StreamGeneration) { + if g.state == nil { + return + } + a.mu.Lock() + defer a.mu.Unlock() + if !g.state.streamCompleted { + return + } + g.state.streamCompleted = false + g.state.streamCancelReason = "delivery_cancelled" + a.publishStreamReleaseLocked(g.state, g.state.streamCancelReason) +} + +// PendingStreamBatches returns every decoded batch of the generation whose +// range starts at or after fromStart, in shred-index order. It reads the +// prefetch state directly, so it is the authoritative recovery path after a +// dropped wake-up. A completed generation still owns its immutable ready +// results, so completion does not hide batches behind queued/lost notifications. +// Cancelled generations return nothing. No new prefetch work is scheduled here. +func (a *SlotAssembler) PendingStreamBatches(g StreamGeneration, fromStart uint32) []*StreamBatch { + if g.state == nil { + return nil + } + a.mu.Lock() + defer a.mu.Unlock() + status := a.streamStatusLocked(g) + if status == StreamGone || g.state.prefetch == nil || (g.state.prefetch.released && status != StreamDone) { + return nil + } + var out []*StreamBatch + for start, batch := range g.state.prefetch.batches { + if start < fromStart { + continue + } + select { + case <-batch.ready: + default: + continue + } + out = append(out, newStreamBatch(g, batch)) + } + sort.Slice(out, func(i, j int) bool { return out[i].Start < out[j].Start }) + return out +} + +// newStreamBatch builds the immutable view; it must only be called after the +// batch's ready channel closed (its fields are immutable from then on). +func newStreamBatch(g StreamGeneration, batch *prefetchedShredBatch) *StreamBatch { + batch.viewOnce.Do(func() { + batch.view = buildStreamBatch(g, batch) + }) + return batch.view +} + +func buildStreamBatch(g StreamGeneration, batch *prefetchedShredBatch) *StreamBatch { + readyAt := batch.readyAt + if readyAt.IsZero() { // detached test batches have no prefetch publication + readyAt = time.Now() + } + view := &StreamBatch{ + Slot: g.slot, + Generation: g, + Start: batch.start, + End: batch.end, + Err: batch.err, + ReadyAt: readyAt, + batch: batch, + } + switch { + case batch.err != nil: + case batch.marker && batch.parent != nil: + view.ParentSlot = batch.parent.ParentSlot + view.ParentBlockID = batch.parent.ParentBlockID + view.ReplayFECSetIndex = batch.parent.ReplayFECSetIndex + if batch.parent.FromUpdateParent { + view.Marker = StreamMarkerUpdateParent + } else { + view.Marker = StreamMarkerHeader + } + case batch.marker && batch.footer != nil: + view.Marker = StreamMarkerFooter + case batch.marker: + // A marker without decoded content is treated like a footer-less + // component boundary: nothing to execute, nothing to select. + view.Marker = StreamMarkerFooter + default: + view.Transactions = batch.transactions + } + return view +} + +// publishStreamBatchReady is called by the prefetch worker, under the +// assembler lock, after the batch's ready channel closed. +func (a *SlotAssembler) publishStreamBatchReadyLocked(s *slotState, batch *prefetchedShredBatch) { + if a.streamSubscriber == nil || s == nil || batch == nil { + return + } + a.noteChildRepairHeaderLocked(s, batch) + a.publishStreamLocked(StreamEvent{Kind: StreamBatchReady, Slot: s.slot, state: weak.Make(s), batch: weak.Make(batch)}) +} + +// publishStreamReleaseLocked is called from releasePrefetchLocked, i.e. from +// every path that drops a slot generation, and tells the subscriber whether a +// complete block follows (finalizeCompletion marked it) or the state is gone. +func (a *SlotAssembler) publishStreamReleaseLocked(s *slotState, reason string) { + if a.streamSubscriber == nil || s == nil { + return + } + state := weak.Make(s) + if s.streamCompleted { + a.publishStreamLocked(StreamEvent{Kind: StreamCompleted, Slot: s.slot, state: state}) + return + } + a.publishStreamLocked(StreamEvent{Kind: StreamCancelled, Slot: s.slot, state: state, Reason: reason}) +} diff --git a/pkg/turbine/stream_test.go b/pkg/turbine/stream_test.go new file mode 100644 index 000000000..bb0277916 --- /dev/null +++ b/pkg/turbine/stream_test.go @@ -0,0 +1,323 @@ +package turbine + +import ( + "context" + "runtime" + "sync" + "sync/atomic" + "testing" + "time" + "weak" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func nextStreamEvent(t *testing.T, ch <-chan StreamEvent, kind StreamEventKind) StreamEvent { + t.Helper() + deadline := time.After(3 * time.Second) + for { + select { + case event := <-ch: + event, live := event.Resolve() + if !live { + continue + } + if event.Kind == kind { + return event + } + case <-deadline: + t.Fatalf("no stream event of kind %d", kind) + } + } +} + +// The feed publishes each prefetched batch once it is decoded — the header +// marker with its parent identity, then entry batches whose transactions are +// the very objects the completed block references — and ends with a +// completion event for the same generation. The final component (the ending +// tick) is decoded by completion, never by the prefetch, so it is not fed. +func TestStreamFeedPublishesBatchesAndCompletion(t *testing.T) { + // The production verifier (nil hook) is the one that attaches message + // identities; a per-transaction hook verifies without producing them. + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + events := make(chan StreamEvent, 64) + a.SubscribeStream(events) + + const slot = 300 + parentID := solana.Hash{9, 9, 9} + batches := prefetchTestShreds(t, slot, + testAlpenglowParentMarkerBytes(blockMarkerVariantHeader, slot-1, parentID), + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), + prefetchTestPayload(t, verifierSignedTransactions(t, 4)), + buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + header := nextStreamEvent(t, events, StreamBatchReady) + require.Equal(t, uint64(slot), header.Slot) + require.False(t, header.Generation.IsZero()) + require.Equal(t, StreamMarkerHeader, header.Batch.Marker) + require.Equal(t, uint64(slot-1), header.Batch.ParentSlot) + require.Equal(t, parentID, header.Batch.ParentBlockID) + require.Empty(t, header.Batch.Transactions) + _, verified, err := header.Batch.WaitVerification(context.Background()) + require.NoError(t, err) + require.False(t, verified, "markers carry no verification") + + require.Nil(t, feedPrefetchShreds(t, a, batches[1])) + first := nextStreamEvent(t, events, StreamBatchReady) + require.Equal(t, header.Generation, first.Generation) + require.Equal(t, batches[1][0].Index, first.Batch.Start) + require.Equal(t, StreamMarkerNone, first.Batch.Marker) + require.Len(t, first.Batch.Transactions, 3) + require.NoError(t, first.Batch.Err) + require.Equal(t, StreamActive, a.StreamStatusOf(first.Generation)) + + identities, verified, err := first.Batch.WaitVerification(context.Background()) + require.NoError(t, err) + require.True(t, verified) + require.Len(t, identities, 3) + prepared, err := block.PrepareVerifiedTransactionMessageIdentities(first.Batch.Transactions, identities) + require.NoError(t, err) + require.Equal(t, 3, prepared.Len()) + + // Recovery path: the pending list must show the same batches by range. + pending := a.PendingStreamBatches(first.Generation, 0) + require.Len(t, pending, 2) + require.Equal(t, header.Batch.Start, pending[0].Start) + require.Equal(t, first.Batch.Start, pending[1].Start) + require.Equal(t, first.Batch.End, pending[1].End) + require.Empty(t, a.PendingStreamBatches(first.Generation, first.Batch.End+1)) + + require.Nil(t, feedPrefetchShreds(t, a, batches[2])) + second := nextStreamEvent(t, events, StreamBatchReady) + require.Equal(t, first.Generation, second.Generation) + require.Len(t, second.Batch.Transactions, 4) + // Make sure the prefetch has retained both entry batches before the last + // component completes the slot; completion then reuses them by identity. + waitPrefetchedBatch(t, a, slot, second.Batch.Start) + + blk := feedPrefetchShreds(t, a, batches[3]) + require.NotNil(t, blk) + done := nextStreamEvent(t, events, StreamCompleted) + require.Equal(t, first.Generation, done.Generation) + require.Equal(t, StreamDone, a.StreamStatusOf(first.Generation)) + retained := a.PendingStreamBatches(first.Generation, first.Batch.Start) + require.Len(t, retained, 2, "completion preserves already-ready entry batches") + for i, batch := range retained { + ids, ok, err := batch.WaitVerification(context.Background()) + require.NoError(t, err) + require.True(t, ok) + require.Len(t, ids, len(batch.Transactions)) + offset := 0 + if i == 1 { + offset = 3 + } + for j, tx := range batch.Transactions { + require.Same(t, blk.Transactions[offset+j], tx) + } + } + + // Pointer identity: the prefix a streaming consumer executed is the block. + require.Len(t, blk.Transactions, 7) + for i, tx := range first.Batch.Transactions { + require.Same(t, tx, blk.Transactions[i]) + } + for i, tx := range second.Batch.Transactions { + require.Same(t, tx, blk.Transactions[3+i]) + } + require.Equal(t, uint64(slot-1), blk.SourceParentSlot) + require.True(t, blk.HasAlpenglowParentBlockID) + require.Equal(t, parentID, solana.Hash(blk.AlpenglowParentBlockID)) + require.Zero(t, a.StreamDroppedEvents()) +} + +// A reset while a slot is streaming cancels the generation; re-assembling the +// slot produces a different generation. +func TestStreamFeedCancelsOnResetAndRenewsGeneration(t *testing.T) { + v := newTransactionVerifier(2, 16, func(*solana.Transaction) error { return nil }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + events := make(chan StreamEvent, 64) + a.SubscribeStream(events) + + const slot = 301 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 2)), + prefetchTestPayload(t, verifierSignedTransactions(t, 2))) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + first := nextStreamEvent(t, events, StreamBatchReady) + + a.ResetSlot(slot) + cancelled := nextStreamEvent(t, events, StreamCancelled) + require.Equal(t, first.Generation, cancelled.Generation) + require.Equal(t, "reset", cancelled.Reason) + require.Equal(t, StreamGone, a.StreamStatusOf(first.Generation)) + require.Empty(t, a.PendingStreamBatches(first.Generation, 0)) + + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + renewed := nextStreamEvent(t, events, StreamBatchReady) + require.NotEqual(t, first.Generation, renewed.Generation) + require.Equal(t, StreamActive, a.StreamStatusOf(renewed.Generation)) +} + +// Dropped wake-ups are counted and never lose state: the batches remain +// discoverable through PendingStreamBatches. +func TestStreamFeedDropsWakeupsWhenSubscriberIsFull(t *testing.T) { + v := newTransactionVerifier(2, 16, func(*solana.Transaction) error { return nil }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + events := make(chan StreamEvent) // unbuffered and never drained: every send drops + a.SubscribeStream(events) + + const slot = 302 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 2)), + prefetchTestPayload(t, verifierSignedTransactions(t, 2))) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, slot, 0) + require.NotNil(t, cached) + require.Eventually(t, func() bool { return a.StreamDroppedEvents() >= 1 }, 3*time.Second, time.Millisecond) + + a.mu.Lock() + state := a.slots[slot] + a.mu.Unlock() + g := StreamGeneration{slot: slot, state: state} + pending := a.PendingStreamBatches(g, 0) + require.Len(t, pending, 1) + require.Len(t, pending[0].Transactions, 2) +} + +// Notifications must not own retired slots or decoded payloads. Conversely, +// resolving a live event gives the consumer a strong polling handle. +func TestStreamQueuedEventsDoNotRetainRetiredBuffers(t *testing.T) { + events := make(chan StreamEvent, 4) + publish := func() (weak.Pointer[slotState], weak.Pointer[prefetchedShredBatch]) { + a := NewSlotAssembler() + a.SubscribeStream(events) + batch := &prefetchedShredBatch{raw: make([]byte, 1<<20), marker: true} + state := &slotState{slot: 42, prefetch: &slotEntryPrefetch{batches: map[uint32]*prefetchedShredBatch{0: batch}}} + a.mu.Lock() + a.publishStreamBatchReadyLocked(state, batch) + a.publishStreamReleaseLocked(state, "reset") + a.mu.Unlock() + return weak.Make(state), weak.Make(batch) + } + state, batch := publish() + require.Eventually(t, func() bool { + runtime.GC() + return state.Value() == nil && batch.Value() == nil + }, 3*time.Second, time.Millisecond) + require.Len(t, events, 2) + for len(events) > 0 { + _, live := (<-events).Resolve() + require.False(t, live) + } +} + +func TestStreamResolvedEventRetainsCompletedPollingState(t *testing.T) { + a := NewSlotAssembler() + events := make(chan StreamEvent, 2) + a.SubscribeStream(events) + ready := make(chan struct{}) + close(ready) + batch := &prefetchedShredBatch{ready: ready, marker: true} + state := &slotState{slot: 42, streamCompleted: true, prefetch: &slotEntryPrefetch{batches: map[uint32]*prefetchedShredBatch{0: batch}, released: true}} + a.mu.Lock() + a.publishStreamBatchReadyLocked(state, batch) + a.publishStreamReleaseLocked(state, "") + a.mu.Unlock() + event, live := (<-events).Resolve() + require.True(t, live) + runtime.GC() + require.Equal(t, StreamDone, a.StreamStatusOf(event.Generation)) + require.Len(t, a.PendingStreamBatches(event.Generation, 0), 1) + done, live := (<-events).Resolve() + require.True(t, live) + require.Equal(t, event.Generation, done.Generation) + require.Equal(t, StreamCompleted, done.Kind) +} + +// The streaming observer must return while the owning request is still running. +// Completion/cleanup retain the separate joining wait and own buffer lifetime. +func TestStreamVerificationTimeoutDoesNotCancelOrJoinOwner(t *testing.T) { + done := make(chan struct{}) + var closeOnce sync.Once + defer closeOnce.Do(func() { close(done) }) + var cancelled atomic.Bool + future := &transactionVerification{done: done, cancel: func() { cancelled.Store(true) }} + batch := &StreamBatch{batch: &prefetchedShredBatch{verification: future}} + ctx, cancel := context.WithTimeout(context.Background(), time.Millisecond) + defer cancel() + returned := make(chan error, 1) + go func() { _, _, err := batch.WaitVerification(ctx); returned <- err }() + select { + case err := <-returned: + require.ErrorIs(t, err, context.DeadlineExceeded) + case <-time.After(3 * time.Second): + t.Fatal("observer waited for unfinished owner") + } + require.False(t, cancelled.Load(), "observer must not cancel completion's shared work") + closeOnce.Do(func() { close(done) }) + _, verified, err := batch.WaitVerification(context.Background()) + require.NoError(t, err) + require.True(t, verified, "same request remains usable after the observer leaves") +} + +func TestStreamPollingReusesTransactionView(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + batches := prefetchTestShreds(t, 812, prefetchTestPayload(t, verifierSignedTransactions(t, 3)), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + waitPrefetchedBatch(t, a, 812, 0) + a.mu.Lock() + g := StreamGeneration{slot: 812, state: a.slots[812]} + a.mu.Unlock() + first, second := a.PendingStreamBatches(g, 0), a.PendingStreamBatches(g, 0) + require.Len(t, first, 1) + require.Len(t, second, 1) + require.Len(t, first[0].Transactions, 3) + require.Same(t, &first[0].Transactions[0], &second[0].Transactions[0]) +} + +func TestCompletedStreamCancelledWhenDeliveryAbandoned(t *testing.T) { + for _, queued := range []bool{false, true} { + a := &SlotAssembler{slots: make(map[uint64]*slotState)} + events := make(chan StreamEvent, 2) + a.SubscribeStream(events) + g := NewDetachedStreamGeneration(101) + g.state.streamCompleted = true + replacement := NewDetachedStreamGeneration(101) + a.slots[101] = replacement.state + r := &UDPReceiver{assembler: a, blocks: make(chan *block.Block), pendingBlocks: make(map[uint64]int)} + ctx, cancel := context.WithCancel(context.Background()) + cancel() + result := slotCompletionResult{block: &block.Block{Slot: 101}, generation: g, pending: true} + r.startPendingBlock(101) + if queued { + results := make(chan slotCompletionResult, 1) + results <- result + close(results) + r.consumeCompletionResults(ctx, results) + } else { + require.False(t, r.handleCompletionResult(ctx, result)) + } + require.Equal(t, StreamGone, a.StreamStatusOf(g)) + require.Equal(t, StreamActive, a.StreamStatusOf(replacement)) + require.Empty(t, r.pendingBlocks) + event := nextStreamEvent(t, events, StreamCancelled) + require.Equal(t, g, event.Generation) + require.Equal(t, "delivery_cancelled", event.Reason) + } +} diff --git a/pkg/turbine/stream_view_test.go b/pkg/turbine/stream_view_test.go new file mode 100644 index 000000000..8f861fb3c --- /dev/null +++ b/pkg/turbine/stream_view_test.go @@ -0,0 +1,33 @@ +package turbine + +import ( + "sync" + "testing" + "time" +) + +func TestReadyStreamViewReused(t *testing.T) { + g := NewDetachedStreamGeneration(42) + ready := make(chan struct{}) + at := time.Now().Add(-time.Second) + b := &prefetchedShredBatch{ready: ready, readyAt: at, start: 1, end: 2} + close(ready) + expected := newStreamBatch(g, b) + var wg sync.WaitGroup + for i := 0; i < 16; i++ { + wg.Add(1) + go func() { + defer wg.Done() + for j := 0; j < 100; j++ { + got := newStreamBatch(g, b) + if got != expected || !got.ReadyAt.Equal(at) { + t.Error("view or readiness time changed") + } + } + }() + } + wg.Wait() + if n := testing.AllocsPerRun(100, func() { _ = newStreamBatch(g, b) }); n != 0 { + t.Fatalf("cached view allocates: %v", n) + } +} diff --git a/pkg/turbine/streaming_message_identity_test.go b/pkg/turbine/streaming_message_identity_test.go new file mode 100644 index 000000000..4fce6debd --- /dev/null +++ b/pkg/turbine/streaming_message_identity_test.go @@ -0,0 +1,104 @@ +package turbine + +import ( + "context" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestMessageIdentitiesPreparedBeforeFinalShred(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + batches := prefetchTestShreds(t, 100, + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), + prefetchTestPayload(t, verifierSignedTransactions(t, 4))) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, 100, 0) + _, err := cached.verification.wait() + require.NoError(t, err) + require.Len(t, cached.verification.identities, 3) + for i := range cached.entries[0].Txns { + tx := &cached.entries[0].Txns[i] + got, ok := cached.verification.identities[i].ForTransaction(tx) + require.True(t, ok) + want, err := txstatus.IdentityForTransaction(tx) + require.NoError(t, err) + require.Equal(t, want, got) + } + blk := feedPrefetchShreds(t, a, batches[1]) + require.NotNil(t, blk) + prepared, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + for i, tx := range blk.Transactions { + want, err := txstatus.IdentityForTransaction(tx) + require.NoError(t, err) + require.Equal(t, want, prepared.Identity(i)) + } +} + +func TestVerifiedIdentityCacheRejectsMismatchedAndPartialCoverage(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + txs := verifierSignedTransactions(t, 3) + request, err := v.submitTransactions(context.Background(), txs) + require.NoError(t, err) + _, err = request.wait() + require.NoError(t, err) + blk := &block.Block{Transactions: txs} + require.NoError(t, blk.CacheVerifiedTransactionMessageIdentities(request.identities)) + prepared, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + require.Error(t, blk.CacheVerifiedTransactionMessageIdentities(request.identities[:2])) + copyTx := *txs[0] + for _, changed := range [][]*solana.Transaction{{txs[1], txs[0], txs[2]}, {©Tx, txs[1], txs[2]}} { + other := &block.Block{Transactions: changed} + require.Error(t, other.CacheVerifiedTransactionMessageIdentities(request.identities)) + } + again, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + require.Same(t, prepared, again, "failed cache adoption must not replace a valid cache") + txs[0].Message.RecentBlockhash[0] ^= 1 + require.Error(t, blk.CacheVerifiedTransactionMessageIdentities(request.identities)) +} + +func TestMessageIdentityFallbackPreservesRetainedTransactionOrder(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + batches := prefetchTestShreds(t, 100, + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), + prefetchTestPayload(t, verifierSignedTransactions(t, 4)), + prefetchTestPayload(t, verifierSignedTransactions(t, 5))) + for i := 0; i < 2; i++ { + require.Nil(t, feedPrefetchShreds(t, a, batches[i])) + cached := waitPrefetchedBatch(t, a, 100, batches[i][0].Index) + _, err := cached.verification.wait() + require.NoError(t, err) + if i == 1 { + // Force a canceled, joined result in the middle of the retained + // sequence. Completion must reverify and scatter its identities. + done := make(chan struct{}) + close(done) + cached.verification = &transactionVerification{done: done, err: context.Canceled, index: -1, cancel: func() {}} + } + } + blk := feedPrefetchShreds(t, a, batches[2]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 12) + prepared, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + for i, tx := range blk.Transactions { + want, err := txstatus.IdentityForTransaction(tx) + require.NoError(t, err) + require.Equal(t, want, prepared.Identity(i)) + } +} diff --git a/pkg/turbine/transaction_job_groups_test.go b/pkg/turbine/transaction_job_groups_test.go new file mode 100644 index 000000000..ab2edb554 --- /dev/null +++ b/pkg/turbine/transaction_job_groups_test.go @@ -0,0 +1,118 @@ +package turbine + +import ( + "context" + "fmt" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestWideVerificationJobsPreserveSignatureFailureIndex(t *testing.T) { + txs := verifierSignedTransactions(t, 320) + txs[33].Signatures[0][0] ^= 1 + txs[98].Signatures[0][0] ^= 1 + for _, groups := range []int{1, 4, 8} { + v := newTransactionVerifierWithJobGroups(2, 32, 8, groups, nil) + r, err := v.submitTransactions(context.Background(), txs) + require.NoError(t, err) + index, err := r.wait() + require.Error(t, err) + require.Equal(t, 33, index) + v.closeAndWait() + } +} + +func TestWideVerificationJobsYieldToReadySmallRequest(t *testing.T) { + for _, groups := range []int{4, 8} { + t.Run(fmt.Sprint(groups), func(t *testing.T) { + large, small := verifierTestBlock(800), verifierTestBlock(4) + started, release := make(chan struct{}), make(chan struct{}) + var releaseOnce sync.Once + var calls, beforeSmall atomic.Int32 + v := newTransactionVerifierWithJobGroups(1, 16, 8, groups, func(tx *solana.Transaction) error { + for _, s := range small.Transactions { + if tx == s { + beforeSmall.Store(calls.Load()) + return nil + } + } + calls.Add(1) + if tx == large.Transactions[0] { + close(started) + <-release + } + return nil + }) + defer v.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + big, err := v.submitTransactions(context.Background(), large.Transactions) + require.NoError(t, err) + waitSignal(t, started, "large job") + little, err := v.submitTransactions(context.Background(), small.Transactions) + require.NoError(t, err) + require.Eventually(t, func() bool { return len(v.jobs) == 1 }, 3*time.Second, time.Millisecond) + releaseOnce.Do(func() { close(release) }) + _, err = little.wait() + require.NoError(t, err) + require.Equal(t, int32(groups*8), beforeSmall.Load(), "one large job, not an entire catch-up request, precedes small work") + _, err = big.wait() + require.NoError(t, err) + require.Equal(t, int32(800), calls.Load()) + }) + } +} + +func TestWideVerificationJobsCancelBetweenVectorsAndJoin(t *testing.T) { + for _, groups := range []int{4, 8} { + t.Run(fmt.Sprint(groups), func(t *testing.T) { + started, release := make(chan struct{}), make(chan struct{}) + var releaseOnce sync.Once + var calls atomic.Int32 + v := newTransactionVerifierWithJobGroups(1, 16, 8, groups, func(*solana.Transaction) error { + if calls.Add(1) == 1 { + close(started) + <-release + } + return nil + }) + defer v.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + r, err := v.submitTransactions(ctx, verifierTestBlock(800).Transactions) + require.NoError(t, err) + waitSignal(t, started, "first vector") + cancel() + select { + case <-r.done: + t.Fatal("released transactions still read by a worker") + default: + } + releaseOnce.Do(func() { close(release) }) + _, err = r.wait() + require.ErrorIs(t, err, context.Canceled) + require.Equal(t, int32(8), calls.Load(), "canceled large job finishes its admitted vector only") + }) + } +} + +func TestWideVerificationJobsDoNotWaitForFourTransactionBatch(t *testing.T) { + for _, groups := range []int{4, 8} { + v := newTransactionVerifierWithJobGroups(2, 32, 8, groups, func(*solana.Transaction) error { return nil }) + r, err := v.submitTransactions(context.Background(), verifierTestBlock(4).Transactions) + require.NoError(t, err) + select { + case <-r.done: + case <-time.After(3 * time.Second): + t.Fatal("partial work waited for another submission") + } + _, err = r.wait() + require.NoError(t, err) + v.closeAndWait() + } +} diff --git a/pkg/turbine/transaction_verifier.go b/pkg/turbine/transaction_verifier.go index e64e36696..050e126b3 100644 --- a/pkg/turbine/transaction_verifier.go +++ b/pkg/turbine/transaction_verifier.go @@ -3,8 +3,8 @@ package turbine import ( "context" "fmt" - "runtime" "sync" + "time" "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/sigverify" @@ -12,105 +12,340 @@ import ( "github.com/gagliardetto/solana-go" ) -var errNilTransaction = fmt.Errorf("nil transaction") +var ( + errNilTransaction = fmt.Errorf("nil transaction") + errTransactionVerifierClosed = fmt.Errorf("transaction verifier closed") +) + +// Four vector groups amortize dispatch for large ready requests while bounding +// the work that can precede another component on a worker. +const defaultTransactionJobGroups = 4 +// A job is formed before admission, from transactions which are already +// available. Workers never wait for more transactions to fill a vector group. type transactionVerifyJob struct { - tx *solana.Transaction - err *error - done *sync.WaitGroup + trace bool + offeredAt, workerStart, workerEnd int64 + ctx context.Context + txs []*solana.Transaction + identities []txverify.VerifiedMessageIdentity + errs []error + start int + done chan<- *transactionVerifyJob } -type transactionVerifier struct { - jobs chan transactionVerifyJob - verify func(*solana.Transaction) error - workers int - // wave is how many transactions verifyBlockContext admits at once. It is - // workers * sigverify.BatchTarget so each worker can actually accumulate a - // full vector group rather than being handed one transaction at a time. - wave int - close sync.Once +// transactionVerification owns an asynchronous request until done closes. +// Transactions submitted to it must remain immutable until wait returns. +type transactionVerification struct { + trace *entryVerificationTrace + done chan struct{} + cancel context.CancelFunc + index int + err error + finishedAt time.Time + identities []txverify.VerifiedMessageIdentity +} - worker sync.WaitGroup +func (r *transactionVerification) wait() (int, error) { + return r.waitContext(context.Background()) } -func newTransactionVerifier(workers, queueDepth int, verify func(*solana.Transaction) error) *transactionVerifier { - if workers < 1 { - workers = 1 +// waitContext cancels further admission when ctx is canceled, but joins every +// admitted job before returning. The caller may then safely release or mutate +// the transaction objects, including their backing message byte slices. +func (r *transactionVerification) waitContext(ctx context.Context) (int, error) { + if ctx == nil { + ctx = context.Background() + } + select { + case <-r.done: + case <-ctx.Done(): + r.cancel() + <-r.done + return -1, ctx.Err() } - wave := workers * sigverify.BatchTarget - if queueDepth < 1 { - queueDepth = 1 + if err := ctx.Err(); err != nil { + return -1, err } + return r.index, r.err +} + +type transactionVerifier struct { + jobs chan *transactionVerifyJob + verify func(*solana.Transaction) error + workers int + batchTarget int + jobGroups int + // Each accepted request owns at most workers outstanding jobs. The + // admission semaphore also bounds asynchronous request goroutines; callers + // apply backpressure before handing off another decoded component. + requests chan struct{} + // Protected by mu. Reserve one request permit for completion/recovery. + prefetchRequests int + completionWaiters int + admissionChanged chan struct{} + request sync.WaitGroup + mu sync.Mutex + closed bool + stopped chan struct{} + close sync.Once + worker sync.WaitGroup +} + +func newTransactionVerifier(workers, queueDepth int, verify func(*solana.Transaction) error) *transactionVerifier { + return newTransactionVerifierWithBatchTarget(workers, queueDepth, sigverify.BatchTarget, verify) +} + +// queueDepth is a transaction budget, rounded up to whole jobs. batchTarget +// counts signature lanes: multi-signature transactions stay indivisible and +// may exceed the target. Four/eight targets can be compared without changing +// the admission or cancellation policy. +func newTransactionVerifierWithBatchTarget(workers, queueDepth, batchTarget int, verify func(*solana.Transaction) error) *transactionVerifier { + return newTransactionVerifierWithJobGroups(workers, queueDepth, batchTarget, defaultTransactionJobGroups, verify) +} + +// Job groups amortize dispatch over already available vector groups. They do +// not change vector width or wait for future transactions to arrive. +func newTransactionVerifierWithJobGroups(workers, queueDepth, batchTarget, jobGroups int, verify func(*solana.Transaction) error) *transactionVerifier { + workers = max(1, workers) + batchTarget = max(1, min(batchTarget, sigverify.BatchTarget)) + jobGroups = max(1, min(jobGroups, 8)) + jobCapacity := batchTarget * jobGroups + queueGroups := max(1, (queueDepth+jobCapacity-1)/jobCapacity) v := &transactionVerifier{ - jobs: make(chan transactionVerifyJob, queueDepth), - verify: verify, - workers: workers, - wave: wave, + jobs: make(chan *transactionVerifyJob, queueGroups), + verify: verify, + workers: workers, + batchTarget: batchTarget, + jobGroups: jobGroups, + requests: make(chan struct{}, 2*workers), + stopped: make(chan struct{}), } v.worker.Add(workers) for i := 0; i < workers; i++ { go func() { defer v.worker.Done() - // Worker-local scratch reused across groups. - var ( - group []transactionVerifyJob - scr verifyScratch - ) + var batch txverify.BatchVerifier for job := range v.jobs { - group = sigverify.Drain(group, job, v.jobs, - sigverify.FairShare(len(v.jobs), v.workers, sigverify.BatchTarget)) - v.verifyGroup(group, &scr) - // Do not keep finished jobs reachable through the scratch. - clear(group) + v.verifyGroup(job, &batch) } }() } return v } -// verifyScratch is one worker's reusable buffers. -type verifyScratch struct { - txs []*solana.Transaction - errs []error - batch txverify.BatchVerifier -} - -// verifyGroup verifies a drained group and releases every job in it. -// -// Releasing happens in a defer covering the whole group, so no caller can be -// left waiting on a job that was drained into a batch which then failed — -// a stranded job would hang verifyBlockContext's done.Wait() forever. -func (v *transactionVerifier) verifyGroup(group []transactionVerifyJob, scr *verifyScratch) { +// verifyGroup releases its job even if signature verification panics. The +// request's bounded completion channel always has room for every pending job. +func (v *transactionVerifier) verifyGroup(job *transactionVerifyJob, batch *txverify.BatchVerifier) { + if job.trace { + job.workerStart = entryTraceNow() + } defer func() { - for _, job := range group { - job.done.Done() + if job.trace { + job.workerEnd = entryTraceNow() } + job.done <- job }() - - // An injected verifier is a per-transaction function and stays that way; - // only the default path can batch. This seam is used by tests. - if v.verify != nil { - for _, job := range group { - *job.err = verifyTransactionSafely(v.verify, job.tx) + for start := 0; start < len(job.txs); { + // An admitted job always finishes its first vector group, preserving + // ownership/join semantics. Cancellation can skip additional groups. + if err := job.ctx.Err(); start > 0 && err != nil { + for i := start; i < len(job.errs); i++ { + job.errs[i] = err + } + return } - return + end := transactionVerifyGroupEnd(job.txs, start, v.batchTarget) + if v.verify != nil { + for i := start; i < end; i++ { + tx := job.txs[i] + if tx == nil { + job.errs[i] = errNilTransaction + } else { + job.errs[i] = verifyTransactionSafely(v.verify, tx) + } + } + } else { + if job.identities != nil { + verifyBatchWithIdentitiesSafely(batch, job.txs[start:end], job.errs[start:end], job.identities[start:end]) + } else { + verifyBatchSafely(batch, job.txs[start:end], job.errs[start:end]) + } + } + start = end } +} + +// submitTransactions admits one immutable decoded component or complete block. +// Admission is bounded and may block: call it from a decode/completion worker, +// never the UDP reader or while holding the assembler mutex. Cancellation of +// ctx stops further groups but still joins every admitted group. +func (v *transactionVerifier) submitTransactions(ctx context.Context, txs []*solana.Transaction) (*transactionVerification, error) { + return v.submitRequest(ctx, txs, false) +} + +// submitPrefetchTransactions applies backpressure before allocating a request: +// prefetch may use at most 2*workers-1 of the existing 2*workers permits. +func (v *transactionVerifier) submitPrefetchTransactions(ctx context.Context, txs []*solana.Transaction) (*transactionVerification, error) { + return v.submitRequest(ctx, txs, true) +} + +func (v *transactionVerifier) submitRequest(ctx context.Context, txs []*solana.Transaction, prefetch bool) (*transactionVerification, error) { + if ctx == nil { + ctx = context.Background() + } + if err := ctx.Err(); err != nil { + return nil, err + } + var trace *entryVerificationTrace + if entryTraceContext(ctx) { + trace = &entryVerificationTrace{Submit: entryTraceNow(), Transactions: len(txs)} + } + if err := v.acquireRequest(ctx, prefetch); err != nil { + return nil, err + } + ctx, cancel := context.WithCancel(ctx) + if trace != nil { + trace.Admitted = entryTraceNow() + } + r := &transactionVerification{done: make(chan struct{}), cancel: cancel, index: -1, trace: trace} + if v.verify == nil { + r.identities = make([]txverify.VerifiedMessageIdentity, len(txs)) + } + go func() { + defer v.request.Done() + defer v.releaseRequest(prefetch) + defer cancel() + r.index, r.err = v.verifyTransactionsWithTiming(ctx, txs, r.identities, trace) + if trace != nil { + trace.Finished = entryTraceNow() + } + r.finishedAt = time.Now() + close(r.done) + }() + return r, nil +} + +// verifyTransactions keeps a rolling window instead of waiting for an entire +// worker wave. A slow job cannot idle workers whose earlier jobs finished. +// One caller can queue at most workers jobs, so a large catch-up block cannot +// put all its transactions ahead of a newly available component. +func (v *transactionVerifier) verifyTransactions(ctx context.Context, txs []*solana.Transaction) (int, error) { + return v.verifyTransactionsWithIdentities(ctx, txs, nil) +} - scr.txs = scr.txs[:0] - for _, job := range group { - scr.txs = append(scr.txs, job.tx) +func (v *transactionVerifier) verifyTransactionsWithIdentities(ctx context.Context, txs []*solana.Transaction, identities []txverify.VerifiedMessageIdentity) (int, error) { + return v.verifyTransactionsWithTiming(ctx, txs, identities, nil) +} + +func (v *transactionVerifier) verifyTransactionsWithTiming(ctx context.Context, txs []*solana.Transaction, identities []txverify.VerifiedMessageIdentity, trace *entryVerificationTrace) (int, error) { + if len(txs) == 0 { + return -1, ctx.Err() + } + window := min(v.workers, len(txs)) + jobGroups := v.jobGroups + // Keep short components responsive and enough independent jobs to supply + // every worker. This is a ready-work threshold, never a batching timer. + if len(txs) < 2*v.workers*v.batchTarget*jobGroups { + jobGroups = 1 } - if cap(scr.errs) < len(scr.txs) { - scr.errs = make([]error, len(scr.txs)) + jobCapacity := v.batchTarget * jobGroups + completed := make(chan *transactionVerifyJob, window) + groups := make([]transactionVerifyJob, window) + errs := make([]error, window*jobCapacity) + free := make([]*transactionVerifyJob, window) + for i := range groups { + groups[i].trace = trace != nil + groups[i].ctx = ctx + groups[i].errs = errs[i*jobCapacity : (i+1)*jobCapacity] + groups[i].done = completed + free[i] = &groups[i] } - scr.errs = scr.errs[:len(scr.txs)] - verifyBatchSafely(&scr.batch, scr.txs, scr.errs) + nextIndex, active := 0, 0 + failureIndex := -1 + var failure error + var pending *transactionVerifyJob + ctxDone := ctx.Done() + stopped := false + for active > 0 || (!stopped && nextIndex < len(txs)) { + if !stopped && ctx.Err() != nil { + stopped = true + ctxDone = nil + } + if stopped && active == 0 { + break + } + if !stopped && pending == nil && nextIndex < len(txs) && len(free) > 0 { + pending = free[len(free)-1] + free = free[:len(free)-1] + end := nextIndex + for group := 0; group < jobGroups && end < len(txs); group++ { + end = transactionVerifyGroupEnd(txs, end, v.batchTarget) + } + if trace != nil { + pending.offeredAt = entryTraceNow() + } + pending.start = nextIndex + pending.txs = txs[nextIndex:end] + if identities != nil { + pending.identities = identities[nextIndex:end] + } + pending.errs = pending.errs[:end-nextIndex] + clear(pending.errs) + } + var admission chan *transactionVerifyJob + if !stopped && pending != nil { + admission = v.jobs + } + select { + case admission <- pending: + nextIndex += len(pending.txs) + active++ + pending = nil + case job := <-completed: + if trace != nil { + trace.observe(job) + } + active-- + for i, err := range job.errs { + if err != nil && (failureIndex < 0 || job.start+i < failureIndex) { + failureIndex, failure = job.start+i, err + stopped = true + } + } + job.txs = nil + job.identities = nil + clear(job.errs) + free = append(free, job) + case <-ctxDone: + stopped = true + ctxDone = nil + } + } + if err := ctx.Err(); err != nil { + return -1, err + } + return failureIndex, failure +} - for i, job := range group { - *job.err = scr.errs[i] +func transactionVerifyGroupEnd(txs []*solana.Transaction, start, target int) int { + end, signatures := start, 0 + for end < len(txs) && end-start < target { + count := 1 + if txs[end] != nil { + count = max(1, len(txs[end].Signatures)) + } + if end > start && signatures+count > target { + break + } + signatures += count + end++ + if signatures >= target { + break + } } - clear(scr.txs) + return end } // verifyBatchSafely mirrors verifyTransactionSafely: a panic in the verifier @@ -129,11 +364,28 @@ func verifyBatchSafely(batch *txverify.BatchVerifier, txs []*solana.Transaction, batch.Verify(txs, errs) } +func verifyBatchWithIdentitiesSafely(batch *txverify.BatchVerifier, txs []*solana.Transaction, errs []error, identities []txverify.VerifiedMessageIdentity) { + defer func() { + if recovered := recover(); recovered != nil { + clear(identities) + for i := range errs { + errs[i] = fmt.Errorf("signature verifier panic: %v", recovered) + } + } + }() + batch.VerifyWithMessageIdentities(txs, errs, identities) +} + func (v *transactionVerifier) closeAndWait() { if v == nil { return } v.close.Do(func() { + v.mu.Lock() + v.closed = true + close(v.stopped) + v.mu.Unlock() + v.request.Wait() close(v.jobs) v.worker.Wait() }) @@ -162,47 +414,18 @@ func (v *transactionVerifier) verifyBlockContext(ctx context.Context, blk *block if blk == nil || len(blk.Transactions) == 0 { return nil } - // Admit one worker-wave at a time. A monster block still occupies every - // verifier lane, but cannot park tens of thousands of jobs ahead of a newly - // completed small block in the shared bounded queue. - for chunkStart := 0; chunkStart < len(blk.Transactions); chunkStart += v.wave { - if err := ctx.Err(); err != nil { - return err - } - chunkEnd := min(chunkStart+v.wave, len(blk.Transactions)) - errs := make([]error, chunkEnd-chunkStart) - var done sync.WaitGroup - for txIdx := chunkStart; txIdx < chunkEnd; txIdx++ { - if err := ctx.Err(); err != nil { - done.Wait() - return err - } - tx := blk.Transactions[txIdx] - errIdx := txIdx - chunkStart - if tx == nil { - errs[errIdx] = errNilTransaction - continue - } - done.Add(1) - select { - case v.jobs <- transactionVerifyJob{tx: tx, err: &errs[errIdx], done: &done}: - case <-ctx.Done(): - done.Done() - done.Wait() - return ctx.Err() - } - } - done.Wait() - if err := ctx.Err(); err != nil { - return err - } - for errIdx, err := range errs { - if err != nil { - return formatTransactionVerificationError(blk, chunkStart+errIdx, err) - } - } + request, err := v.submitTransactions(ctx, blk.Transactions) + if err != nil { + return err + } + index, err := request.wait() + if err != nil && index >= 0 { + return formatTransactionVerificationError(blk, index, err) } - return nil + if err == nil && request.identities != nil { + return blk.CacheVerifiedTransactionMessageIdentities(request.identities) + } + return err } func formatTransactionVerificationError(blk *block.Block, txIdx int, err error) error { @@ -226,12 +449,17 @@ var ( defaultTransactionVerifier *transactionVerifier ) -func validateBlockTransactionsContext(ctx context.Context, blk *block.Block) error { +func getDefaultTransactionVerifier() *transactionVerifier { defaultTransactionVerifierOnce.Do(func() { - workers := max(1, (runtime.GOMAXPROCS(0)+1)/2) - defaultTransactionVerifier = newTransactionVerifier(workers, 2*workers*sigverify.BatchTarget, nil) + workers := sigverify.TransactionWorkers() + target := sigverify.TransactionBatchTarget() + defaultTransactionVerifier = newTransactionVerifierWithBatchTarget(workers, 2*workers*target, target, nil) }) - return defaultTransactionVerifier.verifyBlockContext(ctx, blk) + return defaultTransactionVerifier +} + +func validateBlockTransactionsContext(ctx context.Context, blk *block.Block) error { + return getDefaultTransactionVerifier().verifyBlockContext(ctx, blk) } func validateBlockTransactions(blk *block.Block) error { diff --git a/pkg/turbine/transaction_verifier_admission.go b/pkg/turbine/transaction_verifier_admission.go new file mode 100644 index 000000000..c0a20ed2c --- /dev/null +++ b/pkg/turbine/transaction_verifier_admission.go @@ -0,0 +1,76 @@ +package turbine + +import "context" + +// acquireRequest keeps the total request/job bounds unchanged while reserving +// one permit for completion or full-block recovery. Waiting completions win the +// next available permit over prefetch; completions are otherwise equal priority. +// This is not replay-head scheduling: unfinished prefetch already admitted for +// the head keeps its existing rolling job window, and future-slot completions +// also use the reservation. No worker is reserved and no admitted job is evicted. +// +// Prefetch can wait while completions remain queued. It is speculative work and +// resumes when the completion backlog drains. Cancellation/close wake waiters +// without admitting a request; accepted requests retain the full join contract. +func (v *transactionVerifier) acquireRequest(ctx context.Context, prefetch bool) error { + v.mu.Lock() + waitingCompletion := false + defer func() { + if waitingCompletion { + v.completionWaiters-- + v.wakeAdmissionLocked() + } + v.mu.Unlock() + }() + for { + if v.closed { + return errTransactionVerifierClosed + } + if err := ctx.Err(); err != nil { + return err + } + if len(v.requests) < cap(v.requests) && + (!prefetch || (v.prefetchRequests < cap(v.requests)-1 && v.completionWaiters == 0)) { + v.requests <- struct{}{} + if prefetch { + v.prefetchRequests++ + } + // Add under the same lock as close, so closeAndWait cannot finish + // while an accepted request has yet to start its goroutine. + v.request.Add(1) + return nil + } + if !prefetch && !waitingCompletion { + v.completionWaiters++ + waitingCompletion = true + } + if v.admissionChanged == nil { + v.admissionChanged = make(chan struct{}) + } + changed := v.admissionChanged + v.mu.Unlock() + select { + case <-changed: + case <-ctx.Done(): + case <-v.stopped: + } + v.mu.Lock() + } +} + +func (v *transactionVerifier) releaseRequest(prefetch bool) { + v.mu.Lock() + <-v.requests + if prefetch { + v.prefetchRequests-- + } + v.wakeAdmissionLocked() + v.mu.Unlock() +} + +func (v *transactionVerifier) wakeAdmissionLocked() { + if v.admissionChanged != nil { + close(v.admissionChanged) + v.admissionChanged = nil + } +} diff --git a/pkg/turbine/transaction_verifier_admission_benchmark_test.go b/pkg/turbine/transaction_verifier_admission_benchmark_test.go new file mode 100644 index 000000000..9971224f9 --- /dev/null +++ b/pkg/turbine/transaction_verifier_admission_benchmark_test.go @@ -0,0 +1,118 @@ +package turbine + +import ( + "context" + "fmt" + "testing" + "time" + + "github.com/gagliardetto/solana-go" +) + +// Compare the request-class policy with the same workers, rolling job window, +// Narya backend and total work. Shared submits prefetch as ordinary completion +// requests to reproduce the old four-permit occupancy; reserved labels it as +// prefetch. This is a saturation microbenchmark, not observed live p99 or true +// replay-head scheduling. All signatures are verified and every request joined. +func BenchmarkVerifierCompletionReservation(b *testing.B) { + flowConfigureBackend(b) + txs := make([]*solana.Transaction, 4096) + for i := range txs { + txs[i] = flowGeneratedTransaction(b, 228, uint64(i)) + } + for _, size := range []int{256, 4096} { + for _, reserved := range []bool{false, true} { + b.Run(fmt.Sprintf("prefetch_%d/reserved_%t", size, reserved), func(b *testing.B) { + v := newTransactionVerifierWithBatchTarget(2, 32, 8, nil) + defer v.closeAndWait() + submit := v.submitTransactions + if reserved { + submit = v.submitPrefetchTransactions + } + var admission, finish, total []time.Duration + occupied := 0 + ctx := context.WithValue(context.Background(), entryTraceContextKey{}, true) + warm, err := v.submitTransactions(context.Background(), txs[:256]) + if err != nil { + b.Fatal(err) + } + if _, err = warm.wait(); err != nil { + b.Fatal(err) + } + b.ResetTimer() + for range b.N { + start := time.Now() + prior := make([]*transactionVerification, 0, 4) + for range 3 { + r, err := submit(context.Background(), txs[:size]) + if err != nil { + b.Fatal(err) + } + prior = append(prior, r) + } + type result struct { + r *transactionVerification + err error + } + fourth := make(chan result, 1) + attempting := make(chan struct{}) + go func() { + close(attempting) + r, err := submit(context.Background(), txs[:size]) + fourth <- result{r, err} + }() + <-attempting + if !reserved { + last := <-fourth + if last.err != nil { + b.Fatal(last.err) + } + prior = append(prior, last.r) + } + expectedActive := cap(v.requests) + if reserved { + expectedActive-- + } + if len(v.requests) == expectedActive { + occupied++ + } + r, err := v.submitTransactions(ctx, txs[:32]) + if err != nil { + b.Fatal(err) + } + if _, err = r.wait(); err != nil { + b.Fatal(err) + } + admission = append(admission, time.Duration(r.trace.Admitted-r.trace.Submit)) + finish = append(finish, time.Duration(r.trace.Finished-r.trace.Submit)) + if reserved { + last := <-fourth + if last.err != nil { + b.Fatal(last.err) + } + prior = append(prior, last.r) + } + for _, r := range prior { + if _, err = r.wait(); err != nil { + b.Fatal(err) + } + } + total = append(total, time.Since(start)) + } + b.StopTimer() + b.ReportMetric(float64(occupied)/float64(b.N), "occupied_fraction") + for _, metric := range []struct { + name string + values []time.Duration + }{ + {"completion_admit", admission}, {"completion_done", finish}, {"all_work", total}, + } { + flowReportPercentiles(b, metric.values, metric.name) + index := max(0, (len(metric.values)*99+99)/100-1) + b.ReportMetric(float64(metric.values[index])/float64(time.Millisecond), metric.name+"_p99-ms") + b.ReportMetric(float64(metric.values[len(metric.values)-1])/float64(time.Millisecond), metric.name+"_max-ms") + } + }) + } + } +} diff --git a/pkg/turbine/transaction_verifier_admission_test.go b/pkg/turbine/transaction_verifier_admission_test.go new file mode 100644 index 000000000..eef2c9604 --- /dev/null +++ b/pkg/turbine/transaction_verifier_admission_test.go @@ -0,0 +1,172 @@ +package turbine + +import ( + "context" + "sync" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestPrefetchLeavesCompletionPermitAndCancellationJoins(t *testing.T) { + release := make(chan struct{}) + v := newTransactionVerifier(2, 32, func(*solana.Transaction) error { <-release; return nil }) + defer v.closeAndWait() + defer close(release) + for range 3 { + _, err := v.submitPrefetchTransactions(context.Background(), verifierTestBlock(64).Transactions) + require.NoError(t, err) + } + require.Equal(t, 3, len(v.requests)) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + blocked := make(chan error, 1) + go func() { _, err := v.submitPrefetchTransactions(ctx, verifierTestBlock(1).Transactions); blocked <- err }() + accepted := make(chan *transactionVerification, 1) + go func() { + r, err := v.submitTransactions(context.Background(), verifierTestBlock(1).Transactions) + if err != nil { + t.Error(err) + } + accepted <- r + }() + select { + case r := <-accepted: + require.NotNil(t, r) + case <-time.After(3 * time.Second): + t.Fatal("prefetch occupied the reserved completion permit") + } + require.Equal(t, cap(v.requests), len(v.requests), "total bound must not increase") + cancel() + select { + case err := <-blocked: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(3 * time.Second): + t.Fatal("canceled prefetch admission did not return") + } + // The admitted jobs are still reading transactions until release closes. + require.Equal(t, 4, len(v.requests)) +} + +// Exercise permit arbitration without depending on cryptographic job duration. +// Holding permits models admitted requests; each release also joins its request. +func TestCompletionWinsAdmissionAndPrefetchResumes(t *testing.T) { + v := newTransactionVerifier(1, 8, nil) + ctx, cancel := context.WithCancel(context.Background()) + var cleanup sync.WaitGroup + defer v.closeAndWait() + defer cleanup.Wait() + defer cancel() + release := func(prefetch bool) { v.releaseRequest(prefetch); v.request.Done() } + for range 2 { + require.NoError(t, v.acquireRequest(ctx, false)) + } + firstReleased := false + defer func() { + if !firstReleased { + release(false) + } + release(false) + }() + completionAdmitted := make(chan struct{}) + completionRelease := make(chan struct{}) + var once sync.Once + defer once.Do(func() { close(completionRelease) }) + cleanup.Go(func() { + if err := v.acquireRequest(ctx, false); err != nil { + return + } + close(completionAdmitted) + select { + case <-completionRelease: + case <-ctx.Done(): + } + release(false) + }) + require.Eventually(t, func() bool { v.mu.Lock(); defer v.mu.Unlock(); return v.completionWaiters == 1 }, 3*time.Second, time.Millisecond) + prefetchAdmitted := make(chan struct{}) + cleanup.Go(func() { + if err := v.acquireRequest(ctx, true); err != nil { + return + } + close(prefetchAdmitted) + release(true) + }) + release(false) + firstReleased = true + waitSignal(t, completionAdmitted, "priority completion admission") + select { + case <-prefetchAdmitted: + t.Fatal("prefetch bypassed waiting completion") + default: + } + once.Do(func() { close(completionRelease) }) + waitSignal(t, prefetchAdmitted, "prefetch resumed after completion") +} + +func TestCanceledCompletionDoesNotBlockPrefetch(t *testing.T) { + v := newTransactionVerifier(1, 8, nil) + defer v.closeAndWait() + for range 2 { + require.NoError(t, v.acquireRequest(context.Background(), false)) + } + defer func() { + for range 2 { + v.releaseRequest(false) + v.request.Done() + } + }() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + done := make(chan error, 1) + go func() { done <- v.acquireRequest(ctx, false) }() + require.Eventually(t, func() bool { v.mu.Lock(); defer v.mu.Unlock(); return v.completionWaiters == 1 }, 3*time.Second, time.Millisecond) + cancel() + select { + case err := <-done: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(3 * time.Second): + t.Fatal("completion cancellation stranded admission") + } + v.mu.Lock() + require.Zero(t, v.completionWaiters) + v.mu.Unlock() +} + +func TestCloseWakesBothAdmissionClassesAndJoinsAcceptedWork(t *testing.T) { + release := make(chan struct{}) + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { <-release; return nil }) + var once sync.Once + defer v.closeAndWait() + defer once.Do(func() { close(release) }) + _, err := v.submitPrefetchTransactions(context.Background(), verifierTestBlock(1).Transactions) + require.NoError(t, err) + _, err = v.submitTransactions(context.Background(), verifierTestBlock(1).Transactions) + require.NoError(t, err) + results := make(chan error, 2) + for _, prefetch := range []bool{true, false} { + go func() { + _, err := v.submitRequest(context.Background(), verifierTestBlock(1).Transactions, prefetch) + results <- err + }() + } + closed := make(chan struct{}) + go func() { v.closeAndWait(); close(closed) }() + for range 2 { + select { + case err := <-results: + require.ErrorIs(t, err, errTransactionVerifierClosed) + case <-time.After(3 * time.Second): + t.Fatal("close stranded admission") + } + } + select { + case <-closed: + t.Fatal("close returned while jobs still own transactions") + default: + } + once.Do(func() { close(release) }) + waitSignal(t, closed, "close joined accepted work") +} diff --git a/pkg/turbine/transaction_verifier_flow_benchmark_test.go b/pkg/turbine/transaction_verifier_flow_benchmark_test.go new file mode 100644 index 000000000..0c8ce131b --- /dev/null +++ b/pkg/turbine/transaction_verifier_flow_benchmark_test.go @@ -0,0 +1,479 @@ +package turbine + +import ( + "context" + "crypto/ed25519" + "encoding/base64" + "encoding/binary" + "encoding/json" + "fmt" + "os" + "path/filepath" + "sort" + "strconv" + "sync" + "syscall" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/sigverify" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// BenchmarkTransactionVerificationFlow measures the real verifier pool under +// simulated transaction availability, not actual network reception or replay. +// Decode and fixture generation are outside the timer. "tip" spreads complete +// components over 200 ms; this is an explicit workload model, not a claim about +// the cluster's observed component sizes or arrival distribution. No component +// waits for another component to fill a verification batch. +// +// Optional environment: +// +// MITHRIL_SIGVERIFY_FLOW_BACKEND=r51 (use -run '^$' in a fresh test process) +// MITHRIL_SIGVERIFY_FLOW_FIXTURES=/path/to/fixtures (block-*.json, base64 txs) +// MITHRIL_SIGVERIFY_FLOW_COUNT=33760 (generated count or captured prefix limit) +// +// Without captured fixtures, use reproducible signed 228-byte and 1232-byte +// legacy memo transactions. They are signature workloads, not replay fixtures. +// Use -benchtime=3x or another fixed count when comparing configurations so the +// deliberate arrival waits do not change the number of observations. +func BenchmarkTransactionVerificationFlow(b *testing.B) { + flowConfigureBackend(b) + for _, fixture := range flowBenchmarkFixtures(b) { + b.Run(fixture.name, func(b *testing.B) { + for _, workers := range []int{2, 4} { + b.Run(fmt.Sprintf("workers_%d", workers), func(b *testing.B) { + for _, target := range []int{4, 8} { + b.Run(fmt.Sprintf("target_%d", target), func(b *testing.B) { + for _, scenario := range flowScenarios(fixture.blk) { + b.Run(scenario.name, func(b *testing.B) { + flowRunBenchmark(b, workers, target, fixture.blk, scenario) + }) + } + }) + } + }) + } + }) + } +} + +type flowFixture struct { + name string + blk *block.Block +} + +type flowScenario struct { + name string + components [][]*solana.Transaction + arrivalSpan time.Duration + overlap bool +} + +func flowScenarios(blk *block.Block) []flowScenario { + // 60 KiB is a benchmark parameter only. Include signed transaction bytes; + // actual serialized entry/component overhead and shred recovery are omitted. + components := flowComponents(blk.Transactions, 60*1024) + scenarios := []flowScenario{ + {name: "catchup", components: [][]*solana.Transaction{blk.Transactions}}, + {name: "tip_200ms_after_complete", components: components, arrivalSpan: 200 * time.Millisecond}, + {name: "tip_200ms_overlap", components: components, arrivalSpan: 200 * time.Millisecond, overlap: true}, + } + // Sparse components expose tail behavior without requiring microsecond + // timer precision to deliver tens of thousands of tiny arrival events. + for _, width := range []int{4, 7, 8} { + txs := blk.Transactions[:min(256, len(blk.Transactions))] + var small [][]*solana.Transaction + for start := 0; start < len(txs); start += width { + small = append(small, txs[start:min(start+width, len(txs))]) + } + scenarios = append(scenarios, flowScenario{ + name: fmt.Sprintf("sparse_%dtx_200ms_overlap", width), components: small, + arrivalSpan: 200 * time.Millisecond, overlap: true, + }) + } + return scenarios +} + +func flowComponents(txs []*solana.Transaction, bytesPerComponent int) [][]*solana.Transaction { + var components [][]*solana.Transaction + start, size := 0, 0 + for i, tx := range txs { + wireSize, err := txverify.TransactionWireSize(tx) + if err != nil { + panic(err) // fixtures are validated before entering this helper + } + if i > start && size+wireSize > bytesPerComponent { + components = append(components, txs[start:i]) + start, size = i, 0 + } + size += wireSize + } + if start < len(txs) { + components = append(components, txs[start:]) + } + return components +} + +type flowObservation struct { + latencies []time.Duration + submits []time.Duration + feedLags []time.Duration + residual time.Duration + err error +} + +func flowArrivalOffset(index, count int, span time.Duration) time.Duration { + if count <= 1 { + return span + } + return time.Duration(int64(span) * int64(index) / int64(count-1)) +} + +func flowObserve(v *transactionVerifier, blk *block.Block, scenario flowScenario) flowObservation { + observation := flowObservation{ + latencies: make([]time.Duration, len(scenario.components)), + submits: make([]time.Duration, len(scenario.components)), + feedLags: make([]time.Duration, len(scenario.components)), + } + started := time.Now() + finalArrival := started.Add(scenario.arrivalSpan) + if !scenario.overlap { + time.Sleep(time.Until(finalArrival)) + submitStarted := time.Now() + future, err := v.submitTransactions(context.Background(), blk.Transactions) + submitDuration := time.Since(submitStarted) + if err != nil { + observation.err = err + return observation + } + _, observation.err = future.wait() + finished := future.finishedAt + for i := range observation.latencies { + available := started.Add(flowArrivalOffset(i, len(scenario.components), scenario.arrivalSpan)) + observation.latencies[i] = finished.Sub(available) + observation.submits[i] = submitDuration + observation.feedLags[i] = submitStarted.Sub(available) + } + observation.residual = max(0, finished.Sub(finalArrival)) + return observation + } + + var waiters sync.WaitGroup + errs := make([]error, len(scenario.components)) + finished := make([]time.Time, len(scenario.components)) + for i, txs := range scenario.components { + available := started.Add(flowArrivalOffset(i, len(scenario.components), scenario.arrivalSpan)) + time.Sleep(time.Until(available)) + submitStarted := time.Now() + observation.feedLags[i] = submitStarted.Sub(available) + future, err := v.submitTransactions(context.Background(), txs) + observation.submits[i] = time.Since(submitStarted) + if err != nil { + errs[i] = err + break + } + waiters.Add(1) + go func() { + defer waiters.Done() + _, errs[i] = future.wait() + finished[i] = future.finishedAt + observation.latencies[i] = finished[i].Sub(available) + }() + } + waiters.Wait() + for _, completed := range finished { + observation.residual = max(observation.residual, completed.Sub(finalArrival)) + } + for _, err := range errs { + if err != nil { + observation.err = err + break + } + } + return observation +} + +func flowRunBenchmark(b *testing.B, workers, target int, blk *block.Block, scenario flowScenario) { + v := newTransactionVerifierWithBatchTarget(workers, 2*workers*8, target, nil) + defer v.closeAndWait() + if err := v.verifyBlock(blk); err != nil { + b.Fatal(err) + } + flowBenchmarkGate(b) + var signatureCount int + for _, component := range scenario.components { + for _, tx := range component { + signatureCount += len(tx.Signatures) + } + } + var latencies, submits, feedLags, residuals []time.Duration + before := sigverify.Stats() + cpuBefore := flowCPUSeconds(b) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + observation := flowObserve(v, blk, scenario) + if observation.err != nil { + b.Fatal(observation.err) + } + latencies = append(latencies, observation.latencies...) + submits = append(submits, observation.submits...) + feedLags = append(feedLags, observation.feedLags...) + residuals = append(residuals, observation.residual) + } + b.StopTimer() + cpuSeconds := flowCPUSeconds(b) - cpuBefore + after := sigverify.Stats() + b.ReportMetric(1000*cpuSeconds/float64(b.N), "cpu-ms/block") + b.ReportMetric(cpuSeconds/b.Elapsed().Seconds(), "avg_cpu_cores") + b.ReportMetric(float64(b.N*signatureCount)/b.Elapsed().Seconds(), "signatures/s") + b.ReportMetric(float64(signatureCount), "signatures/block") + b.ReportMetric(float64(len(scenario.components)), "components/block") + b.ReportMetric(float64(after.Signatures-before.Signatures)/float64(after.Batches-before.Batches), "mean_width") + flowReportPercentiles(b, latencies, "ready") + flowReportPercentiles(b, submits, "submit") + flowReportPercentiles(b, feedLags, "feed_lag") + flowReportPercentiles(b, residuals, "residual") + if after.InternalFaultFallbacks != before.InternalFaultFallbacks { + b.Fatal("signature verifier used an internal fault fallback") + } + if want := uint64(b.N * signatureCount); after.Signatures-before.Signatures != want { + b.Fatalf("verified signature count = %d, want %d", after.Signatures-before.Signatures, want) + } +} + +// The optional rendezvous is outside the timer. An external contention runner +// starts the real execution probe after seeing READY, then creates START. Both +// paths are explicit files in that runner's output directory. This is only for +// coordination; ordinary benchmark invocations do no filesystem polling. +func flowBenchmarkGate(b *testing.B) { + b.Helper() + ready, start := os.Getenv("MITHRIL_SIGVERIFY_FLOW_READY"), os.Getenv("MITHRIL_SIGVERIFY_FLOW_START") + if ready == "" && start == "" { + return + } + if ready == "" || start == "" { + b.Fatal("set both MITHRIL_SIGVERIFY_FLOW_READY and MITHRIL_SIGVERIFY_FLOW_START") + } + // Go's first N=1 calibration and warmup must finish before the external + // execution probe starts. The runner selects a fixed N greater than one. + if b.N == 1 { + return + } + if err := os.WriteFile(ready, []byte(b.Name()+"\n"), 0600); err != nil { + b.Fatal(err) + } + deadline := time.Now().Add(30 * time.Second) + for { + if _, err := os.Stat(start); err == nil { + return + } else if !os.IsNotExist(err) { + b.Fatal(err) + } + if time.Now().After(deadline) { + b.Fatal("contention runner did not release benchmark within 30 seconds") + } + time.Sleep(time.Millisecond) + } +} + +func flowReportPercentiles(b *testing.B, values []time.Duration, prefix string) { + sort.Slice(values, func(i, j int) bool { return values[i] < values[j] }) + for _, percentile := range []int{50, 95} { + index := max(0, (len(values)*percentile+99)/100-1) + b.ReportMetric(float64(values[index])/float64(time.Millisecond), fmt.Sprintf("%s_p%d-ms", prefix, percentile)) + } +} + +func flowCPUSeconds(tb testing.TB) float64 { + tb.Helper() + var usage syscall.Rusage + if err := syscall.Getrusage(syscall.RUSAGE_SELF, &usage); err != nil { + tb.Fatal(err) + } + return float64(usage.Utime.Sec+usage.Stime.Sec) + float64(usage.Utime.Usec+usage.Stime.Usec)/1e6 +} + +var flowBackendOnce sync.Once +var flowBackendError error + +func flowConfigureBackend(tb testing.TB) { + tb.Helper() + flowBackendOnce.Do(func() { + if backend := os.Getenv("MITHRIL_SIGVERIFY_FLOW_BACKEND"); backend != "" { + _, flowBackendError = sigverify.Configure(sigverify.Config{Backend: backend}) + } + }) + if flowBackendError != nil { + tb.Fatal(flowBackendError) + } +} + +func flowBenchmarkFixtures(tb testing.TB) []flowFixture { + tb.Helper() + count := 33760 + limit := false + if value := os.Getenv("MITHRIL_SIGVERIFY_FLOW_COUNT"); value != "" { + var err error + count, err = strconv.Atoi(value) + if err != nil || count < 1 { + tb.Fatal("MITHRIL_SIGVERIFY_FLOW_COUNT must be a positive integer") + } + limit = true + } + if dir := os.Getenv("MITHRIL_SIGVERIFY_FLOW_FIXTURES"); dir != "" { + paths, err := filepath.Glob(filepath.Join(dir, "block-*.json")) + if err != nil || len(paths) == 0 { + tb.Fatalf("captured fixtures: %v, files=%d", err, len(paths)) + } + var fixtures []flowFixture + for _, path := range paths { + data, err := os.ReadFile(path) + if err != nil { + tb.Fatal(err) + } + var captured struct { + Slot uint64 + Transactions []string + } + if err := json.Unmarshal(data, &captured); err != nil { + tb.Fatal(err) + } + blk := &block.Block{Slot: captured.Slot} + for i, encoded := range captured.Transactions { + if limit && i == count { + break + } + wire, err := base64.StdEncoding.DecodeString(encoded) + if err != nil { + tb.Fatal(err) + } + tx, err := solana.TransactionFromBytes(wire) + if err != nil { + tb.Fatal(err) + } + if err := txverify.SanitizeTransaction(tx); err != nil { + tb.Fatal(err) + } + blk.Transactions = append(blk.Transactions, tx) + } + if len(blk.Transactions) == 0 { + tb.Fatalf("fixture %s contains no transactions", path) + } + fixtures = append(fixtures, flowFixture{fmt.Sprintf("captured_%d", blk.Slot), blk}) + } + return fixtures + } + var fixtures []flowFixture + for _, wireSize := range []int{228, txverify.MaxLegacyTransactionSize} { + blk := &block.Block{Slot: 1, Transactions: make([]*solana.Transaction, count)} + for i := range blk.Transactions { + blk.Transactions[i] = flowGeneratedTransaction(tb, wireSize, uint64(i)) + } + fixtures = append(fixtures, flowFixture{fmt.Sprintf("generated_%dB", wireSize), blk}) + } + return fixtures +} + +func flowGeneratedTransaction(tb testing.TB, wireSize int, index uint64) *solana.Transaction { + tb.Helper() + // Public, deterministic benchmark material, never a validator identity. + var seed [ed25519.SeedSize]byte + binary.LittleEndian.PutUint64(seed[:], index+1) + private := ed25519.NewKeyFromSeed(seed[:]) + public := solana.PublicKeyFromBytes(private.Public().(ed25519.PublicKey)) + tx := &solana.Transaction{ + Signatures: make([]solana.Signature, 1), + Message: solana.Message{ + Header: solana.MessageHeader{NumRequiredSignatures: 1, NumReadonlyUnsignedAccounts: 1}, + AccountKeys: []solana.PublicKey{public, solana.MemoProgramID}, + Instructions: []solana.CompiledInstruction{{ProgramIDIndex: 1, Accounts: []uint16{0}}}, + }, + } + binary.LittleEndian.PutUint64(tx.Message.RecentBlockhash[:], index+1) + baseSize, err := txverify.TransactionWireSize(tx) + if err != nil { + tb.Fatal(err) + } + padding := wireSize - baseSize + for attempts := 0; attempts < 3 && padding >= 0; attempts++ { + tx.Message.Instructions[0].Data = make([]byte, padding) + size, err := txverify.TransactionWireSize(tx) + if err != nil { + tb.Fatal(err) + } + if size != wireSize { + padding += wireSize - size + continue + } + for i := range tx.Message.Instructions[0].Data { + tx.Message.Instructions[0].Data[i] = 'a' + } + message, err := txverify.MessageBytes(tx) + if err != nil { + tb.Fatal(err) + } + copy(tx.Signatures[0][:], ed25519.Sign(private, message)) + if err := txverify.SanitizeTransaction(tx); err != nil { + tb.Fatal(err) + } + return tx + } + tb.Fatalf("cannot construct a %d-byte transaction", wireSize) + return nil +} + +func TestTransactionVerificationFlowFixtureShape(t *testing.T) { + for _, size := range []int{228, 1232} { + first := flowGeneratedTransaction(t, size, 0) + second := flowGeneratedTransaction(t, size, 1) + for _, tx := range []*solana.Transaction{first, second} { + wire, err := tx.MarshalBinary() + if err != nil { + t.Fatal(err) + } + if len(wire) != size || len(tx.Signatures) != 1 { + t.Fatalf("fixture wire=%d signatures=%d, want %d bytes and one signature", len(wire), len(tx.Signatures), size) + } + if err := txverify.VerifyTransaction(tx); err != nil { + t.Fatal(err) + } + } + if first.Signatures[0] == second.Signatures[0] || first.Message.AccountKeys[0] == second.Message.AccountKeys[0] { + t.Fatal("different fixture indices reused a message signature or signer") + } + } +} + +func TestTransactionVerificationFlowComponentBoundaries(t *testing.T) { + txs := make([]*solana.Transaction, 7) + for i := range txs { + txs[i] = flowGeneratedTransaction(t, 228, uint64(i)) + } + components := flowComponents(txs, 500) + if len(components) != 4 { + t.Fatalf("got %d components, want four", len(components)) + } + next := 0 + for i, component := range components { + want := 2 + if i == 3 { + want = 1 + } + if len(component) != want { + t.Fatalf("component %d contains %d transactions, want %d", i, len(component), want) + } + for _, tx := range component { + if tx != txs[next] { + t.Fatal("component split changed transaction order or identity") + } + next++ + } + } + if flowArrivalOffset(0, 4, 200*time.Millisecond) != 0 || flowArrivalOffset(3, 4, 200*time.Millisecond) != 200*time.Millisecond { + t.Fatal("availability schedule does not span the requested interval") + } +} diff --git a/pkg/turbine/transaction_verifier_test.go b/pkg/turbine/transaction_verifier_test.go index 8eb19c63d..dbb73e859 100644 --- a/pkg/turbine/transaction_verifier_test.go +++ b/pkg/turbine/transaction_verifier_test.go @@ -1,9 +1,12 @@ package turbine import ( + "context" + "crypto/ed25519" "errors" "fmt" "strings" + "sync" "sync/atomic" "testing" "time" @@ -43,12 +46,12 @@ func TestTransactionVerifierBoundsConcurrencyAndQueue(t *testing.T) { return nil }) defer verifier.closeAndWait() - if cap(verifier.jobs) != 2*workers { - t.Fatalf("queue capacity = %d, want %d", cap(verifier.jobs), 2*workers) + if got, want := cap(verifier.jobs), 1; got != want { + t.Fatalf("group queue capacity = %d, want %d", got, want) } done := make(chan error, 1) - go func() { done <- verifier.verifyBlock(verifierTestBlock(12)) }() + go func() { done <- verifier.verifyBlock(verifierTestBlock(24)) }() deadline := time.After(3 * time.Second) for active.Load() != workers { select { @@ -76,7 +79,7 @@ func TestTransactionVerifierBoundsConcurrencyAndQueue(t *testing.T) { } func TestTransactionVerifierReturnsLowestFailingIndex(t *testing.T) { - blk := verifierTestBlock(6) + blk := verifierTestBlock(24) lowErr := errors.New("low index failure") highErr := errors.New("high index failure") verifier := newTransactionVerifier(4, 8, func(tx *solana.Transaction) error { @@ -84,7 +87,7 @@ func TestTransactionVerifierReturnsLowestFailingIndex(t *testing.T) { case blk.Transactions[1]: time.Sleep(10 * time.Millisecond) return lowErr - case blk.Transactions[3]: + case blk.Transactions[9]: return highErr default: return nil @@ -132,8 +135,8 @@ func TestTransactionVerifierRejectsNilAtDeterministicIndex(t *testing.T) { // Every transaction in a block must be verified and joined, whatever the count. // Workers group transactions, so a count that divides badly into groups must -// not leave a remainder waiting for company: verifyBlockContext joins each -// wave with done.Wait(), and a stranded job would hang it forever. +// not leave a remainder waiting for company: every tail is dispatched as +// soon as it is available, without waiting for another request. // // A counting verifier is injected so the assertion is on what was actually // verified, not merely on returning without error. @@ -161,3 +164,290 @@ func TestTransactionVerifierVerifiesEveryTransactionForAwkwardCounts(t *testing. }) } } + +func TestTransactionVerifierRefillsWhileEarlierGroupIsBlocked(t *testing.T) { + blk := verifierTestBlock(24) + started := make(chan struct{}) + release := make(chan struct{}) + refilled := make(chan struct{}) + v := newTransactionVerifierWithBatchTarget(2, 16, 8, func(tx *solana.Transaction) error { + switch tx { + case blk.Transactions[0]: + close(started) + <-release + case blk.Transactions[16]: + close(refilled) + } + return nil + }) + defer v.closeAndWait() + defer close(release) + r, err := v.submitTransactions(context.Background(), blk.Transactions) + require.NoError(t, err) + waitSignal(t, started, "slow first group") + waitSignal(t, refilled, "rolling refill before first group finishes") + select { + case <-r.done: + t.Fatal("request finished without joining its blocked group") + default: + } +} + +func TestTransactionVerifierPartialTailStartsWithoutAnotherSubmission(t *testing.T) { + for _, target := range []int{4, 8} { + t.Run(fmt.Sprintf("target=%d", target), func(t *testing.T) { + seen := make(chan struct{}, 3) + v := newTransactionVerifierWithBatchTarget(2, 16, target, func(*solana.Transaction) error { + seen <- struct{}{} + return nil + }) + defer v.closeAndWait() + r, err := v.submitTransactions(context.Background(), verifierTestBlock(3).Transactions) + require.NoError(t, err) + for range 3 { + waitSignal(t, seen, "available partial batch transaction") + } + _, err = r.wait() + require.NoError(t, err) + require.False(t, r.finishedAt.IsZero()) + }) + } +} + +func TestTransactionVerifierLargeRequestDoesNotQueuePastSmallRequest(t *testing.T) { + large := verifierTestBlock(800) + small := verifierTestBlock(1) + started := make(chan struct{}) + release := make(chan struct{}) + var releaseOnce sync.Once + var largeCalls atomic.Int32 + var callsBeforeSmall atomic.Int32 + v := newTransactionVerifierWithBatchTarget(1, 16, 8, func(tx *solana.Transaction) error { + if tx == small.Transactions[0] { + callsBeforeSmall.Store(largeCalls.Load()) + return nil + } + largeCalls.Add(1) + if tx == large.Transactions[0] { + close(started) + <-release + } + return nil + }) + defer v.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + largeRequest, err := v.submitTransactions(context.Background(), large.Transactions) + require.NoError(t, err) + waitSignal(t, started, "large request first group") + smallRequest, err := v.submitTransactions(context.Background(), small.Transactions) + require.NoError(t, err) + require.Eventually(t, func() bool { return len(v.jobs) == 1 }, 3*time.Second, time.Millisecond) + releaseOnce.Do(func() { close(release) }) + _, err = smallRequest.wait() + require.NoError(t, err) + require.Equal(t, int32(v.batchTarget*v.jobGroups), callsBeforeSmall.Load(), "large request may only stay one job ahead") + _, err = largeRequest.wait() + require.NoError(t, err) + require.Equal(t, int32(800), largeCalls.Load()) +} + +func TestTransactionVerifierAsyncAdmissionAppliesCancelableBackpressure(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + var first sync.Once + v := newTransactionVerifierWithBatchTarget(1, 8, 8, func(*solana.Transaction) error { + first.Do(func() { close(started) }) + <-release + return nil + }) + defer v.closeAndWait() + defer close(release) + for range 2 { + _, err := v.submitTransactions(context.Background(), verifierTestBlock(1).Transactions) + require.NoError(t, err) + } + waitSignal(t, started, "occupied request slots") + require.Equal(t, cap(v.requests), len(v.requests)) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + result := make(chan error, 1) + go func() { + _, err := v.submitTransactions(ctx, verifierTestBlock(1).Transactions) + result <- err + }() + select { + case err := <-result: + t.Fatalf("unbounded request admitted instead of waiting: %v", err) + case <-time.After(20 * time.Millisecond): + } + cancel() + select { + case err := <-result: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(3 * time.Second): + t.Fatal("request admission ignored cancellation") + } +} + +func TestTransactionVerificationWaitContextCancelsAndJoins(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + var releaseOnce sync.Once + var seen atomic.Int32 + v := newTransactionVerifierWithBatchTarget(1, 8, 8, func(*solana.Transaction) error { + if seen.Add(1) == 1 { + close(started) + <-release + } + return nil + }) + defer v.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + r, err := v.submitTransactions(context.Background(), verifierTestBlock(800).Transactions) + require.NoError(t, err) + waitSignal(t, started, "first admitted group") + ctx, cancel := context.WithCancel(context.Background()) + cancel() + result := make(chan error, 1) + go func() { _, err := r.waitContext(ctx); result <- err }() + select { + case err := <-result: + t.Fatalf("wait returned while transactions still owned by worker: %v", err) + case <-time.After(20 * time.Millisecond): + } + releaseOnce.Do(func() { close(release) }) + select { + case err := <-result: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(3 * time.Second): + t.Fatal("canceled request did not join") + } + require.Equal(t, int32(8), seen.Load(), "cancellation must stop later group admission") +} + +func TestTransactionVerifierCloseRacesAdmissionWithoutStrandingRequests(t *testing.T) { + v := newTransactionVerifier(2, 16, func(*solana.Transaction) error { return nil }) + var callers sync.WaitGroup + for range 32 { + callers.Go(func() { + r, err := v.submitTransactions(context.Background(), verifierTestBlock(17).Transactions) + if err != nil { + if !errors.Is(err, errTransactionVerifierClosed) { + t.Errorf("submit error: %v", err) + } + return + } + _, err = r.wait() + if err != nil { + t.Errorf("admitted request error: %v", err) + } + }) + } + v.closeAndWait() + callers.Wait() + _, err := v.submitTransactions(context.Background(), verifierTestBlock(1).Transactions) + require.ErrorIs(t, err, errTransactionVerifierClosed) +} + +func verifierSignedTransactions(t *testing.T, count int) []*solana.Transaction { + t.Helper() + seed := make([]byte, ed25519.SeedSize) + seed[0] = 71 // Deterministic test-only key; never a validator identity. + key := ed25519.NewKeyFromSeed(seed) + var public solana.PublicKey + copy(public[:], key[32:]) + txs := make([]*solana.Transaction, count) + for i := range txs { + tx := &solana.Transaction{ + Message: solana.Message{ + Header: solana.MessageHeader{NumRequiredSignatures: 1}, + AccountKeys: []solana.PublicKey{public}, + RecentBlockhash: solana.Hash{byte(i), byte(i >> 8)}, + }, + Signatures: make([]solana.Signature, 1), + } + message, err := tx.Message.MarshalBinary() + require.NoError(t, err) + copy(tx.Signatures[0][:], ed25519.Sign(key, message)) + txs[i] = tx + } + return txs +} + +func TestTransactionVerifierRejectsEveryInvalidSignatureLane(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + for invalid := range 8 { + t.Run(fmt.Sprintf("lane=%d", invalid), func(t *testing.T) { + txs := verifierSignedTransactions(t, 8) + txs[invalid].Signatures[0][13] ^= 0x40 + r, err := v.submitTransactions(context.Background(), txs) + require.NoError(t, err) + index, err := r.wait() + require.ErrorContains(t, err, "invalid signature") + require.Equal(t, invalid, index) + }) + } + r, err := v.submitTransactions(context.Background(), verifierSignedTransactions(t, 17)) + require.NoError(t, err) + _, err = r.wait() + require.NoError(t, err, "valid transactions must still verify after invalid lanes") +} + +func TestTransactionVerifierKeepsMultisignatureTransactionsIntactAcrossTargets(t *testing.T) { + counts := []int{2, 2, 1, 4, 9, 3, 2, 1} + txs := make([]*solana.Transaction, len(counts)) + for i, count := range counts { + keys := make([]ed25519.PrivateKey, count) + public := make([]solana.PublicKey, count) + for signer := range keys { + seed := make([]byte, ed25519.SeedSize) + seed[0], seed[1] = 83, byte(signer) + keys[signer] = ed25519.NewKeyFromSeed(seed) + copy(public[signer][:], keys[signer][32:]) + } + tx := &solana.Transaction{ + Message: solana.Message{ + Header: solana.MessageHeader{ + NumRequiredSignatures: uint8(count), + NumReadonlySignedAccounts: uint8(count - 1), + }, + AccountKeys: public, + RecentBlockhash: solana.Hash{byte(i)}, + }, + Signatures: make([]solana.Signature, count), + } + message, err := tx.Message.MarshalBinary() + require.NoError(t, err) + for signer, key := range keys { + copy(tx.Signatures[signer][:], ed25519.Sign(key, message)) + } + txs[i] = tx + } + for _, target := range []int{4, 8} { + for _, groups := range []int{1, 4, 8} { + t.Run(fmt.Sprintf("target=%d/groups=%d", target, groups), func(t *testing.T) { + v := newTransactionVerifierWithJobGroups(2, 16, target, groups, nil) + defer v.closeAndWait() + var large []*solana.Transaction + for range 40 { + large = append(large, txs...) + } + r, err := v.submitTransactions(context.Background(), large) + require.NoError(t, err) + _, err = r.wait() + require.NoError(t, err) + + // Corrupt a non-first signer after an oversized (nine-signature) + // transaction. Results must still map to the original tx index. + txs[6].Signatures[1][11] ^= 0x20 + r, err = v.submitTransactions(context.Background(), large) + require.NoError(t, err) + index, err := r.wait() + require.ErrorContains(t, err, "invalid signature") + require.Equal(t, 6, index) + txs[6].Signatures[1][11] ^= 0x20 + }) + } + } +} diff --git a/pkg/txstatus/message_identity.go b/pkg/txstatus/message_identity.go index 4d41f5450..4541a2a12 100644 --- a/pkg/txstatus/message_identity.go +++ b/pkg/txstatus/message_identity.go @@ -27,11 +27,18 @@ func TransactionMessageHash(tx *solana.Transaction) ([32]byte, error) { return messageHash, fmt.Errorf("serialize transaction message: %w", err) } + return HashCanonicalMessage(message), nil +} + +// HashCanonicalMessage hashes the exact canonical bytes used for transaction +// signature verification, including any message-version prefix. +func HashCanonicalMessage(message []byte) [32]byte { + var messageHash [32]byte hasher := blake3.New() _, _ = hasher.Write([]byte(transactionMessageHashDomain)) _, _ = hasher.Write(message) hasher.Sum(messageHash[:0]) - return messageHash, nil + return messageHash } // IdentityForTransaction captures both components needed for a status-cache diff --git a/pkg/txverify/message_identity.go b/pkg/txverify/message_identity.go new file mode 100644 index 000000000..4ba83bfc2 --- /dev/null +++ b/pkg/txverify/message_identity.go @@ -0,0 +1,25 @@ +package txverify + +import ( + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" +) + +// VerifiedMessageIdentity is an immutable result of signature verification. +// Its zero value is unusable. Signed message contents must remain immutable; +// as with Block's existing cache, arbitrary in-place edits require invalidation. +type VerifiedMessageIdentity struct { + transaction *solana.Transaction + version solana.MessageVersion + identity txstatus.TransactionMessageIdentity + verified bool +} + +// ForTransaction checks the binding without reserializing. Address-table +// resolution is allowed because it does not change the canonical message. +func (v VerifiedMessageIdentity) ForTransaction(tx *solana.Transaction) (txstatus.TransactionMessageIdentity, bool) { + if !v.verified || tx == nil || tx != v.transaction || tx.Message.GetVersion() != v.version || tx.Message.RecentBlockhash != v.identity.RecentBlockhash { + return txstatus.TransactionMessageIdentity{}, false + } + return v.identity, true +} diff --git a/pkg/txverify/message_identity_test.go b/pkg/txverify/message_identity_test.go new file mode 100644 index 000000000..29d265497 --- /dev/null +++ b/pkg/txverify/message_identity_test.go @@ -0,0 +1,70 @@ +package txverify + +import ( + "crypto/ed25519" + "encoding/hex" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func identitySignedTransaction(t *testing.T, version solana.MessageVersion) *solana.Transaction { + t.Helper() + key := ed25519.NewKeyFromSeed(make([]byte, 32)) + tx := &solana.Transaction{Message: solana.Message{ + Header: solana.MessageHeader{NumRequiredSignatures: 1, NumReadonlyUnsignedAccounts: 1}, + AccountKeys: []solana.PublicKey{solana.PublicKeyFromBytes(key.Public().(ed25519.PublicKey)), {2}}, + RecentBlockhash: solana.Hash{3}, + Instructions: []solana.CompiledInstruction{{ProgramIDIndex: 1, Accounts: []uint16{0}, Data: []byte{4}}}, + }} + _, err := tx.Message.SetVersion(version) + require.NoError(t, err) + msg, err := MessageBytes(tx) + require.NoError(t, err) + tx.Signatures = []solana.Signature{solana.SignatureFromBytes(ed25519.Sign(key, msg))} + return tx +} + +func TestVerifiedMessageIdentityCanonicalVersionsAndFailures(t *testing.T) { + wire, err := hex.DecodeString(rustV1FeeHeapTransaction) + require.NoError(t, err) + v1, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + txs := []*solana.Transaction{identitySignedTransaction(t, solana.MessageVersionLegacy), identitySignedTransaction(t, solana.MessageVersionV0), v1} + bad := identitySignedTransaction(t, solana.MessageVersionLegacy) + bad.Signatures[0][0] ^= 1 + txs = append(txs, bad, nil) + errs := make([]error, len(txs)) + ids := make([]VerifiedMessageIdentity, len(txs)) + var verifier BatchVerifier + verifier.VerifyWithMessageIdentities(txs, errs, ids) + for i := range txs { + id, ok := ids[i].ForTransaction(txs[i]) + if i >= 3 { + require.Error(t, errs[i]) + require.False(t, ok) + continue + } + require.NoError(t, errs[i]) + require.True(t, ok) + want, err := txstatus.IdentityForTransaction(txs[i]) + require.NoError(t, err) + require.Equal(t, want, id) + } + // Scratch reuse cannot invalidate a prior successful request, and reusing + // an output lane for failure must not leave a usable old identity behind. + saved := ids[0] + verifier.VerifyWithMessageIdentities([]*solana.Transaction{bad}, errs[:1], ids[:1]) + _, ok := saved.ForTransaction(txs[0]) + require.True(t, ok) + _, ok = ids[0].ForTransaction(txs[0]) + require.False(t, ok) + copyTx := *txs[0] + _, ok = saved.ForTransaction(©Tx) + require.False(t, ok) + txs[0].Message.RecentBlockhash[0] ^= 1 + _, ok = saved.ForTransaction(txs[0]) + require.False(t, ok) +} diff --git a/pkg/txverify/txverify.go b/pkg/txverify/txverify.go index 7221399c4..af58b9cb5 100644 --- a/pkg/txverify/txverify.go +++ b/pkg/txverify/txverify.go @@ -4,6 +4,7 @@ import ( "fmt" "github.com/Overclock-Validator/mithril/pkg/sigverify" + "github.com/Overclock-Validator/mithril/pkg/txstatus" "github.com/gagliardetto/solana-go" ) @@ -222,6 +223,21 @@ type BatchVerifier struct { // Every transaction gets an independent verdict: one bad transaction does not // mask the others, so a caller can report precisely which one failed. func (v *BatchVerifier) Verify(txs []*solana.Transaction, errs []error) { + v.verify(txs, errs, nil) +} + +// VerifyWithMessageIdentities additionally retains identities derived from the +// same canonical bytes used by signature verification. Only successful verdicts +// produce usable identities. The output is caller-owned, not verifier scratch. +func (v *BatchVerifier) VerifyWithMessageIdentities(txs []*solana.Transaction, errs []error, identities []VerifiedMessageIdentity) { + if len(identities) != len(txs) { + panic("txverify: identities and txs length mismatch") + } + clear(identities) + v.verify(txs, errs, identities) +} + +func (v *BatchVerifier) verify(txs []*solana.Transaction, errs []error, identities []VerifiedMessageIdentity) { if len(errs) != len(txs) { panic("txverify: errs and txs length mismatch") } @@ -239,6 +255,16 @@ func (v *BatchVerifier) Verify(txs []*solana.Transaction, errs []error) { v.signers = append(v.signers, nil) continue } + if identities != nil { + identities[i] = VerifiedMessageIdentity{ + transaction: tx, + version: tx.Message.GetVersion(), + identity: txstatus.TransactionMessageIdentity{ + MessageHash: txstatus.HashCanonicalMessage(msg), + RecentBlockhash: tx.Message.RecentBlockhash, + }, + } + } for j := range tx.Signatures { v.batch.Add((*[32]byte)(&signers[j]), msg, tx.Signatures[j][:]) } @@ -247,6 +273,9 @@ func (v *BatchVerifier) Verify(txs []*solana.Transaction, errs []error) { } if v.batch.Verify() { + for i := range identities { + identities[i].verified = errs[i] == nil + } return } @@ -257,6 +286,9 @@ func (v *BatchVerifier) Verify(txs []*solana.Transaction, errs []error) { errs[i] = fmt.Errorf("invalid signature by %s", v.signers[i][j]) } } + if identities != nil { + identities[i].verified = errs[i] == nil + } lane += count } }